CoolFace
Datasetpublic

Anurag1734/cuda-error-resolution-analysis

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes7downloads
topics_batch_184.json58526 linesDownload Raw Back to raw
1[2  {3    "post_stream": {4      "posts": [5        {6          "id": 375483,7          "name": "Florian Bruckner",8          "username": "Florian_Bruckner",9          "avatar_template": "/user_avatar/discuss.pytorch.org/florian_bruckner/{size}/55016_2.png",10          "created_at": "2022-11-19T22:50:36.795Z",11          "cooked": "<p>Hi, I am wondering whether it is possible to make a view of an array with does cyclic permutation or zero padding. With view I mean that no data should be copied.</p>\n<p>This would be very helpful for calculating finite differences with certain boundary conditions.<br>\ntorch.diff allows to append/prepend some data, but for multidimentional arrays it is very cumbersome, and I think that it also copies the data finally.</p>\n<p>Thanks for any advice<br>\nbest wishes<br>\nFlorian</p>",12          "post_number": 1,13          "post_type": 1,14          "posts_count": 1,15          "updated_at": "2022-11-19T22:50:36.795Z",16          "reply_count": 0,17          "reply_to_post_number": null,18          "quote_count": 0,19          "incoming_link_count": 47,20          "reads": 3,21          "readers_count": 2,22          "score": 235.6,23          "yours": false,24          "topic_id": 166419,25          "topic_slug": "views-for-cyclic-permutation-or-zero-padding",26          "display_username": "Florian Bruckner",27          "primary_group_name": null,28          "flair_name": null,29          "flair_url": null,30          "flair_bg_color": null,31          "flair_color": null,32          "flair_group_id": null,33          "badges_granted": [],34          "version": 1,35          "can_edit": false,36          "can_delete": false,37          "can_recover": false,38          "can_see_hidden_post": false,39          "can_wiki": false,40          "read": true,41          "user_title": null,42          "bookmarked": false,43          "actions_summary": [],44          "moderator": false,45          "admin": false,46          "staff": false,47          "user_id": 61120,48          "hidden": false,49          "trust_level": 1,50          "deleted_at": null,51          "user_deleted": false,52          "edit_reason": null,53          "can_view_edit_history": true,54          "wiki": false,55          "post_url": "/t/views-for-cyclic-permutation-or-zero-padding/166419/1",56          "can_accept_answer": false,57          "can_unaccept_answer": false,58          "accepted_answer": false,59          "topic_accepted_answer": null,60          "can_vote": false61        }62      ],63      "stream": [64        37548365      ]66    },67    "timeline_lookup": [68      [69        1,70        107171      ]72    ],73    "suggested_topics": [74      {75        "fancy_title": "OSError: None is not a local folder and is not a valid model identifier listed on &lsquo;https://huggingface.co/models&rsquo;",76        "id": 216993,77        "title": "OSError: None is not a local folder and is not a valid model identifier listed on 'https://huggingface.co/models'",78        "slug": "oserror-none-is-not-a-local-folder-and-is-not-a-valid-model-identifier-listed-on-https-huggingface-co-models",79        "posts_count": 3,80        "reply_count": 1,81        "highest_post_number": 3,82        "image_url": null,83        "created_at": "2025-02-21T15:04:50.224Z",84        "last_posted_at": "2025-02-25T17:32:06.700Z",85        "bumped": true,86        "bumped_at": "2025-02-25T17:32:06.700Z",87        "archetype": "regular",88        "unseen": false,89        "pinned": false,90        "unpinned": null,91        "visible": true,92        "closed": false,93        "archived": false,94        "bookmarked": null,95        "liked": null,96        "tags_descriptions": {},97        "like_count": 0,98        "views": 532,99        "category_id": 1,100        "featured_link": null,101        "has_accepted_answer": true,102        "posters": [103          {104            "extras": "latest",105            "description": "Original Poster, Most Recent Poster",106            "user": {107              "id": 82738,108              "username": "Dmitr",109              "name": "",110              "avatar_template": "/letter_avatar_proxy/v4/letter/d/13edae/{size}.png",111              "trust_level": 1112            }113          },114          {115            "extras": null,116            "description": "Frequent Poster, Accepted Answer",117            "user": {118              "id": 3534,119              "username": "ptrblck",120              "name": "",121              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",122              "admin": true,123              "moderator": true,124              "trust_level": 2125            }126          }127        ]128      },129      {130        "fancy_title": "Hyperparameter Sweeping by Only Measuring Convergence",131        "id": 217603,132        "title": "Hyperparameter Sweeping by Only Measuring Convergence",133        "slug": "hyperparameter-sweeping-by-only-measuring-convergence",134        "posts_count": 1,135        "reply_count": 0,136        "highest_post_number": 1,137        "image_url": null,138        "created_at": "2025-03-08T19:04:40.171Z",139        "last_posted_at": "2025-03-08T19:04:40.209Z",140        "bumped": true,141        "bumped_at": "2025-03-08T19:04:40.209Z",142        "archetype": "regular",143        "unseen": false,144        "pinned": false,145        "unpinned": null,146        "visible": true,147        "closed": false,148        "archived": false,149        "bookmarked": null,150        "liked": null,151        "tags_descriptions": {},152        "like_count": 0,153        "views": 64,154        "category_id": 1,155        "featured_link": null,156        "has_accepted_answer": false,157        "posters": [158          {159            "extras": "latest single",160            "description": "Original Poster, Most Recent Poster",161            "user": {162              "id": 81605,163              "username": "benjamin-perry-duke",164              "name": "Ben Perry",165              "avatar_template": "/user_avatar/discuss.pytorch.org/benjamin-perry-duke/{size}/74629_2.png",166              "trust_level": 1167            }168          }169        ]170      },171      {172        "fancy_title": "Unable to import torch after update no module torch._utils",173        "id": 214243,174        "title": "Unable to import torch after update no module torch._utils",175        "slug": "unable-to-import-torch-after-update-no-module-torch-utils",176        "posts_count": 3,177        "reply_count": 0,178        "highest_post_number": 3,179        "image_url": null,180        "created_at": "2024-12-15T11:34:25.337Z",181        "last_posted_at": "2024-12-19T01:03:21.350Z",182        "bumped": true,183        "bumped_at": "2024-12-19T01:03:21.350Z",184        "archetype": "regular",185        "unseen": false,186        "pinned": false,187        "unpinned": null,188        "visible": true,189        "closed": false,190        "archived": false,191        "bookmarked": null,192        "liked": null,193        "tags_descriptions": {},194        "like_count": 0,195        "views": 305,196        "category_id": 1,197        "featured_link": null,198        "has_accepted_answer": true,199        "posters": [200          {201            "extras": "latest single",202            "description": "Original Poster, Most Recent Poster, Accepted Answer",203            "user": {204              "id": 81514,205              "username": "SpaceBearEngee",206              "name": "SpaceBearEngee",207              "avatar_template": "/user_avatar/discuss.pytorch.org/spacebearengee/{size}/74525_2.png",208              "trust_level": 0209            }210          }211        ]212      },213      {214        "fancy_title": "Troubleshooting LSTM Forecasting Function: What am I doing wrong?",215        "id": 215536,216        "title": "Troubleshooting LSTM Forecasting Function: What am I doing wrong?",217        "slug": "troubleshooting-lstm-forecasting-function-what-am-i-doing-wrong",218        "posts_count": 1,219        "reply_count": 0,220        "highest_post_number": 1,221        "image_url": null,222        "created_at": "2025-01-18T04:26:09.899Z",223        "last_posted_at": "2025-01-18T04:26:09.938Z",224        "bumped": true,225        "bumped_at": "2025-01-18T07:21:34.553Z",226        "archetype": "regular",227        "unseen": false,228        "pinned": false,229        "unpinned": null,230        "visible": true,231        "closed": false,232        "archived": false,233        "bookmarked": null,234        "liked": null,235        "tags_descriptions": {},236        "like_count": 0,237        "views": 26,238        "category_id": 1,239        "featured_link": null,240        "has_accepted_answer": false,241        "posters": [242          {243            "extras": "latest single",244            "description": "Original Poster, Most Recent Poster",245            "user": {246              "id": 82151,247              "username": "NGA",248              "name": "",249              "avatar_template": "/user_avatar/discuss.pytorch.org/nga/{size}/75120_2.png",250              "trust_level": 0251            }252          }253        ]254      },255      {256        "fancy_title": "Help me understand how to fix the error: Process finished with exit code -1073741819 (0xC0000005)",257        "id": 218335,258        "title": "Help me understand how to fix the error: Process finished with exit code -1073741819 (0xC0000005)",259        "slug": "help-me-understand-how-to-fix-the-error-process-finished-with-exit-code-1073741819-0xc0000005",260        "posts_count": 1,261        "reply_count": 0,262        "highest_post_number": 1,263        "image_url": null,264        "created_at": "2025-03-27T12:39:28.760Z",265        "last_posted_at": "2025-03-27T12:39:28.794Z",266        "bumped": true,267        "bumped_at": "2025-03-27T12:39:28.794Z",268        "archetype": "regular",269        "unseen": false,270        "pinned": false,271        "unpinned": null,272        "visible": true,273        "closed": false,274        "archived": false,275        "bookmarked": null,276        "liked": null,277        "tags_descriptions": {},278        "like_count": 0,279        "views": 28,280        "category_id": 1,281        "featured_link": null,282        "has_accepted_answer": false,283        "posters": [284          {285            "extras": "latest single",286            "description": "Original Poster, Most Recent Poster",287            "user": {288              "id": 83484,289              "username": "Behruz_Sherov",290              "name": "Behruz Sherov",291              "avatar_template": "/user_avatar/discuss.pytorch.org/behruz_sherov/{size}/76360_2.png",292              "trust_level": 0293            }294          }295        ]296      }297    ],298    "tags_descriptions": {},299    "fancy_title": "Views for Cyclic Permutation or Zero Padding",300    "id": 166419,301    "title": "Views for Cyclic Permutation or Zero Padding",302    "posts_count": 1,303    "created_at": "2022-11-19T22:50:36.711Z",304    "views": 274,305    "reply_count": 0,306    "like_count": 0,307    "last_posted_at": "2022-11-19T22:50:36.795Z",308    "visible": true,309    "closed": false,310    "archived": false,311    "has_summary": false,312    "archetype": "regular",313    "slug": "views-for-cyclic-permutation-or-zero-padding",314    "category_id": 1,315    "word_count": 78,316    "deleted_at": null,317    "user_id": 61120,318    "featured_link": null,319    "pinned_globally": false,320    "pinned_at": null,321    "pinned_until": null,322    "image_url": null,323    "slow_mode_seconds": 0,324    "draft": null,325    "draft_key": "topic_166419",326    "draft_sequence": null,327    "unpinned": null,328    "pinned": false,329    "current_post_number": 1,330    "highest_post_number": 1,331    "deleted_by": null,332    "actions_summary": [333      {334        "id": 4,335        "count": 0,336        "hidden": false,337        "can_act": false338      },339      {340        "id": 8,341        "count": 0,342        "hidden": false,343        "can_act": false344      },345      {346        "id": 10,347        "count": 0,348        "hidden": false,349        "can_act": false350      },351      {352        "id": 7,353        "count": 0,354        "hidden": false,355        "can_act": false356      }357    ],358    "chunk_size": 20,359    "bookmarked": false,360    "topic_timer": null,361    "message_bus_last_id": 0,362    "participant_count": 1,363    "show_read_indicator": false,364    "thumbnails": null,365    "slow_mode_enabled_until": null,366    "can_vote": false,367    "vote_count": 0,368    "user_voted": false,369    "discourse_zendesk_plugin_zendesk_id": null,370    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",371    "details": {372      "can_edit": false,373      "notification_level": 1,374      "participants": [375        {376          "id": 61120,377          "username": "Florian_Bruckner",378          "name": "Florian Bruckner",379          "avatar_template": "/user_avatar/discuss.pytorch.org/florian_bruckner/{size}/55016_2.png",380          "post_count": 1,381          "primary_group_name": null,382          "flair_name": null,383          "flair_url": null,384          "flair_color": null,385          "flair_bg_color": null,386          "flair_group_id": null,387          "trust_level": 1388        }389      ],390      "created_by": {391        "id": 61120,392        "username": "Florian_Bruckner",393        "name": "Florian Bruckner",394        "avatar_template": "/user_avatar/discuss.pytorch.org/florian_bruckner/{size}/55016_2.png"395      },396      "last_poster": {397        "id": 61120,398        "username": "Florian_Bruckner",399        "name": "Florian Bruckner",400        "avatar_template": "/user_avatar/discuss.pytorch.org/florian_bruckner/{size}/55016_2.png"401      }402    },403    "bookmarks": []404  },405  {406    "post_stream": {407      "posts": [408        {409          "id": 375443,410          "name": "Omega Ma",411          "username": "Omega_Ma",412          "avatar_template": "/user_avatar/discuss.pytorch.org/omega_ma/{size}/52185_2.png",413          "created_at": "2022-11-19T18:23:49.823Z",414          "cooked": "<p>Hi, I have a large-size network that is out of the memory of one GPU. So I put different layers into different GPUs. This has fixed the 'out of memory error when loading the model. However, the error still shows in the step of backpropagation.</p>\n<pre><code class=\"lang-auto\">Traceback (most recent call last):\n  File \"/scratch/project_2005641/THz_DNN/THz_Huge.py\", line 379, in &lt;module&gt;\n    Loss_cache, Lr_list = train_model()\n  File \"/scratch/project_2005641/THz_DNN/THz_Huge.py\", line 127, in train_model\n    loss.backward()  # backpropagation\n  File \"/usr/local/lib64/python3.9/site-packages/torch/_tensor.py\", line 396, in backward\n    torch.autograd.backward(self, gradient, retain_graph, create_graph, inputs=inputs)\n  File \"/usr/local/lib64/python3.9/site-packages/torch/autograd/__init__.py\", line 173, in backward\n    Variable._execution_engine.run_backward(  # Calls into the C++ engine to run the backward pass\nRuntimeError: CUDA out of memory. Tried to allocate 16.00 GiB (GPU 2; 31.75 GiB total capacity; 16.01 GiB already allocated; 14.78 GiB free; 16.02 GiB reserved in total by PyTorch) If reserved memory is &gt;&gt; allocated memory try setting max_split_size_mb to avoid fragmentation.  See documentation for Memory Management and PYTORCH_CUDA_ALLOC_CONF\n</code></pre>\n<p>Any solution to fix this?</p>",415          "post_number": 1,416          "post_type": 1,417          "posts_count": 3,418          "updated_at": "2022-11-19T18:23:49.823Z",419          "reply_count": 0,420          "reply_to_post_number": null,421          "quote_count": 0,422          "incoming_link_count": 804,423          "reads": 9,424          "readers_count": 8,425          "score": 4021.8,426          "yours": false,427          "topic_id": 166409,428          "topic_slug": "cuda-out-of-memory-when-back-propgation",429          "display_username": "Omega Ma",430          "primary_group_name": null,431          "flair_name": null,432          "flair_url": null,433          "flair_bg_color": null,434          "flair_color": null,435          "flair_group_id": null,436          "badges_granted": [],437          "version": 1,438          "can_edit": false,439          "can_delete": false,440          "can_recover": false,441          "can_see_hidden_post": false,442          "can_wiki": false,443          "read": true,444          "user_title": null,445          "bookmarked": false,446          "actions_summary": [],447          "moderator": false,448          "admin": false,449          "staff": false,450          "user_id": 58587,451          "hidden": false,452          "trust_level": 1,453          "deleted_at": null,454          "user_deleted": false,455          "edit_reason": null,456          "can_view_edit_history": true,457          "wiki": false,458          "post_url": "/t/cuda-out-of-memory-when-back-propgation/166409/1",459          "can_accept_answer": false,460          "can_unaccept_answer": false,461          "accepted_answer": false,462          "topic_accepted_answer": null,463          "can_vote": false464        },465        {466          "id": 375470,467          "name": "Chris",468          "username": "pentachris",469          "avatar_template": "/letter_avatar_proxy/v4/letter/p/dec6dc/{size}.png",470          "created_at": "2022-11-19T19:59:11.317Z",471          "cooked": "<p>The backward pass locates additional memory on your GPU to store each parameter’s gradient value. Only leaf tensor nodes (model parameters and inputs) get their gradient stored in the grad attribute. Therefore the memory usage is increasing between the inference (forward pass) and the backpropagation. So its no suprise that you get this message if it even had problems only by inference.<br>\nIam no expert in GPU utilization but you could try:</p>\n<ol>\n<li>Make your Network smaller or try to get a more powerful computer (the obvious ones if its possible :D)</li>\n<li>If you have more GPUs left, try to experiment and use everything you have on more different layers</li>\n<li>You can go to the documentation for memory management and pytorchs cuda config or watch some toutorials/read through the internet about GPU utilization</li>\n</ol>\n<p>Best of luck!</p>",472          "post_number": 2,473          "post_type": 1,474          "posts_count": 3,475          "updated_at": "2022-11-19T19:59:11.317Z",476          "reply_count": 0,477          "reply_to_post_number": null,478          "quote_count": 0,479          "incoming_link_count": 9,480          "reads": 7,481          "readers_count": 6,482          "score": 46.4,483          "yours": false,484          "topic_id": 166409,485          "topic_slug": "cuda-out-of-memory-when-back-propgation",486          "display_username": "Chris",487          "primary_group_name": null,488          "flair_name": null,489          "flair_url": null,490          "flair_bg_color": null,491          "flair_color": null,492          "flair_group_id": null,493          "badges_granted": [],494          "version": 1,495          "can_edit": false,496          "can_delete": false,497          "can_recover": false,498          "can_see_hidden_post": false,499          "can_wiki": false,500          "read": true,501          "user_title": null,502          "bookmarked": false,503          "actions_summary": [],504          "moderator": false,505          "admin": false,506          "staff": false,507          "user_id": 61114,508          "hidden": false,509          "trust_level": 1,510          "deleted_at": null,511          "user_deleted": false,512          "edit_reason": null,513          "can_view_edit_history": true,514          "wiki": false,515          "post_url": "/t/cuda-out-of-memory-when-back-propgation/166409/2",516          "can_accept_answer": false,517          "can_unaccept_answer": false,518          "accepted_answer": false,519          "topic_accepted_answer": null520        },521        {522          "id": 375477,523          "name": "Omega Ma",524          "username": "Omega_Ma",525          "avatar_template": "/user_avatar/discuss.pytorch.org/omega_ma/{size}/52185_2.png",526          "created_at": "2022-11-19T21:21:09.578Z",527          "cooked": "<p>Thanks for the suggestion.</p>",528          "post_number": 3,529          "post_type": 1,530          "posts_count": 3,531          "updated_at": "2022-11-19T21:21:09.578Z",532          "reply_count": 0,533          "reply_to_post_number": null,534          "quote_count": 0,535          "incoming_link_count": 7,536          "reads": 7,537          "readers_count": 6,538          "score": 36.4,539          "yours": false,540          "topic_id": 166409,541          "topic_slug": "cuda-out-of-memory-when-back-propgation",542          "display_username": "Omega Ma",543          "primary_group_name": null,544          "flair_name": null,545          "flair_url": null,546          "flair_bg_color": null,547          "flair_color": null,548          "flair_group_id": null,549          "badges_granted": [],550          "version": 1,551          "can_edit": false,552          "can_delete": false,553          "can_recover": false,554          "can_see_hidden_post": false,555          "can_wiki": false,556          "read": true,557          "user_title": null,558          "bookmarked": false,559          "actions_summary": [],560          "moderator": false,561          "admin": false,562          "staff": false,563          "user_id": 58587,564          "hidden": false,565          "trust_level": 1,566          "deleted_at": null,567          "user_deleted": false,568          "edit_reason": null,569          "can_view_edit_history": true,570          "wiki": false,571          "post_url": "/t/cuda-out-of-memory-when-back-propgation/166409/3",572          "can_accept_answer": false,573          "can_unaccept_answer": false,574          "accepted_answer": false,575          "topic_accepted_answer": null576        }577      ],578      "stream": [579        375443,580        375470,581        375477582      ]583    },584    "timeline_lookup": [585      [586        1,587        1071588      ]589    ],590    "suggested_topics": [591      {592        "fancy_title": "PyTorch for RTX 5090? When will it be out? Thank you",593        "id": 217908,594        "title": "PyTorch for RTX 5090? When will it be out? Thank you",595        "slug": "pytorch-for-rtx-5090-when-will-it-be-out-thank-you",596        "posts_count": 2,597        "reply_count": 0,598        "highest_post_number": 2,599        "image_url": null,600        "created_at": "2025-03-16T10:10:27.595Z",601        "last_posted_at": "2025-03-16T14:00:28.491Z",602        "bumped": true,603        "bumped_at": "2025-03-16T14:00:28.491Z",604        "archetype": "regular",605        "unseen": false,606        "pinned": false,607        "unpinned": null,608        "visible": true,609        "closed": false,610        "archived": false,611        "bookmarked": null,612        "liked": null,613        "tags_descriptions": {},614        "like_count": 0,615        "views": 149,616        "category_id": 1,617        "featured_link": null,618        "has_accepted_answer": false,619        "posters": [620          {621            "extras": null,622            "description": "Original Poster",623            "user": {624              "id": 83305,625              "username": "Raf_Duran",626              "name": "Raf Duran",627              "avatar_template": "/user_avatar/discuss.pytorch.org/raf_duran/{size}/76163_2.png",628              "trust_level": 0629            }630          },631          {632            "extras": "latest",633            "description": "Most Recent Poster",634            "user": {635              "id": 3534,636              "username": "ptrblck",637              "name": "",638              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",639              "admin": true,640              "moderator": true,641              "trust_level": 2642            }643          }644        ]645      },646      {647        "fancy_title": "PyTorch Newbie - Help with Datasets and DatasetLoaders",648        "id": 214541,649        "title": "PyTorch Newbie - Help with Datasets and DatasetLoaders",650        "slug": "pytorch-newbie-help-with-datasets-and-datasetloaders",651        "posts_count": 4,652        "reply_count": 1,653        "highest_post_number": 4,654        "image_url": null,655        "created_at": "2024-12-22T13:31:01.578Z",656        "last_posted_at": "2024-12-23T16:55:44.893Z",657        "bumped": true,658        "bumped_at": "2024-12-23T16:55:44.893Z",659        "archetype": "regular",660        "unseen": false,661        "pinned": false,662        "unpinned": null,663        "visible": true,664        "closed": false,665        "archived": false,666        "bookmarked": null,667        "liked": null,668        "tags_descriptions": {},669        "like_count": 0,670        "views": 50,671        "category_id": 1,672        "featured_link": null,673        "has_accepted_answer": false,674        "posters": [675          {676            "extras": null,677            "description": "Original Poster",678            "user": {679              "id": 81665,680              "username": "irongeek75",681              "name": "",682              "avatar_template": "/user_avatar/discuss.pytorch.org/irongeek75/{size}/74683_2.png",683              "trust_level": 0684            }685          },686          {687            "extras": "latest",688            "description": "Most Recent Poster",689            "user": {690              "id": 3534,691              "username": "ptrblck",692              "name": "",693              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",694              "admin": true,695              "moderator": true,696              "trust_level": 2697            }698          }699        ]700      },701      {702        "fancy_title": "Stack a vector and another tensor of size 0 or 2",703        "id": 214887,704        "title": "Stack a vector and another tensor of size 0 or 2",705        "slug": "stack-a-vector-and-another-tensor-of-size-0-or-2",706        "posts_count": 2,707        "reply_count": 0,708        "highest_post_number": 2,709        "image_url": null,710        "created_at": "2025-01-02T13:41:32.767Z",711        "last_posted_at": "2025-01-02T13:48:14.740Z",712        "bumped": true,713        "bumped_at": "2025-01-02T13:59:44.279Z",714        "archetype": "regular",715        "unseen": false,716        "pinned": false,717        "unpinned": null,718        "visible": true,719        "closed": false,720        "archived": false,721        "bookmarked": null,722        "liked": null,723        "tags_descriptions": {},724        "like_count": 1,725        "views": 30,726        "category_id": 1,727        "featured_link": null,728        "has_accepted_answer": false,729        "posters": [730          {731            "extras": "latest single",732            "description": "Original Poster, Most Recent Poster",733            "user": {734              "id": 78029,735              "username": "AviZ",736              "name": "",737              "avatar_template": "/letter_avatar_proxy/v4/letter/a/e0b2c6/{size}.png",738              "trust_level": 1739            }740          }741        ]742      },743      {744        "fancy_title": "How to calculate the activation memory usage of a model",745        "id": 215402,746        "title": "How to calculate the activation memory usage of a model",747        "slug": "how-to-calculate-the-activation-memory-usage-of-a-model",748        "posts_count": 2,749        "reply_count": 0,750        "highest_post_number": 2,751        "image_url": null,752        "created_at": "2025-01-15T02:31:38.595Z",753        "last_posted_at": "2025-01-16T13:52:14.855Z",754        "bumped": true,755        "bumped_at": "2025-01-16T13:52:14.855Z",756        "archetype": "regular",757        "unseen": false,758        "pinned": false,759        "unpinned": null,760        "visible": true,761        "closed": false,762        "archived": false,763        "bookmarked": null,764        "liked": null,765        "tags_descriptions": {},766        "like_count": 0,767        "views": 185,768        "category_id": 1,769        "featured_link": null,770        "has_accepted_answer": false,771        "posters": [772          {773            "extras": null,774            "description": "Original Poster",775            "user": {776              "id": 72471,777              "username": "shadowshadow",778              "name": "",779              "avatar_template": "/user_avatar/discuss.pytorch.org/shadowshadow/{size}/62985_2.png",780              "trust_level": 2781            }782          },783          {784            "extras": "latest",785            "description": "Most Recent Poster",786            "user": {787              "id": 3534,788              "username": "ptrblck",789              "name": "",790              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",791              "admin": true,792              "moderator": true,793              "trust_level": 2794            }795          }796        ]797      },798      {799        "fancy_title": "NVIDIA GeForce RTX 5070 Ti with CUDA capability sm_120",800        "id": 221509,801        "title": "NVIDIA GeForce RTX 5070 Ti with CUDA capability sm_120",802        "slug": "nvidia-geforce-rtx-5070-ti-with-cuda-capability-sm-120",803        "posts_count": 13,804        "reply_count": 9,805        "highest_post_number": 14,806        "image_url": null,807        "created_at": "2025-07-14T10:11:41.292Z",808        "last_posted_at": "2025-07-17T13:47:45.137Z",809        "bumped": true,810        "bumped_at": "2025-07-17T13:47:45.137Z",811        "archetype": "regular",812        "unseen": false,813        "pinned": false,814        "unpinned": null,815        "visible": true,816        "closed": false,817        "archived": false,818        "bookmarked": null,819        "liked": null,820        "tags_descriptions": {},821        "like_count": 0,822        "views": 4954,823        "category_id": 1,824        "featured_link": null,825        "has_accepted_answer": false,826        "posters": [827          {828            "extras": null,829            "description": "Original Poster",830            "user": {831              "id": 85072,832              "username": "archai",833              "name": "",834              "avatar_template": "/letter_avatar_proxy/v4/letter/a/7cd45c/{size}.png",835              "trust_level": 0836            }837          },838          {839            "extras": null,840            "description": "Frequent Poster",841            "user": {842              "id": 85088,843              "username": "MaxCorvy",844              "name": "Max Corvy",845              "avatar_template": "/user_avatar/discuss.pytorch.org/maxcorvy/{size}/77676_2.png",846              "trust_level": 1847            }848          },849          {850            "extras": "latest",851            "description": "Most Recent Poster",852            "user": {853              "id": 3534,854              "username": "ptrblck",855              "name": "",856              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",857              "admin": true,858              "moderator": true,859              "trust_level": 2860            }861          }862        ]863      }864    ],865    "tags_descriptions": {},866    "fancy_title": "CUDA out of memory when back propgation",867    "id": 166409,868    "title": "CUDA out of memory when back propgation",869    "posts_count": 3,870    "created_at": "2022-11-19T18:23:49.736Z",871    "views": 1256,872    "reply_count": 0,873    "like_count": 0,874    "last_posted_at": "2022-11-19T21:21:09.578Z",875    "visible": true,876    "closed": false,877    "archived": false,878    "has_summary": false,879    "archetype": "regular",880    "slug": "cuda-out-of-memory-when-back-propgation",881    "category_id": 1,882    "word_count": 333,883    "deleted_at": null,884    "user_id": 58587,885    "featured_link": null,886    "pinned_globally": false,887    "pinned_at": null,888    "pinned_until": null,889    "image_url": null,890    "slow_mode_seconds": 0,891    "draft": null,892    "draft_key": "topic_166409",893    "draft_sequence": null,894    "unpinned": null,895    "pinned": false,896    "current_post_number": 1,897    "highest_post_number": 3,898    "deleted_by": null,899    "actions_summary": [900      {901        "id": 4,902        "count": 0,903        "hidden": false,904        "can_act": false905      },906      {907        "id": 8,908        "count": 0,909        "hidden": false,910        "can_act": false911      },912      {913        "id": 10,914        "count": 0,915        "hidden": false,916        "can_act": false917      },918      {919        "id": 7,920        "count": 0,921        "hidden": false,922        "can_act": false923      }924    ],925    "chunk_size": 20,926    "bookmarked": false,927    "topic_timer": null,928    "message_bus_last_id": 0,929    "participant_count": 2,930    "show_read_indicator": false,931    "thumbnails": null,932    "slow_mode_enabled_until": null,933    "can_vote": false,934    "vote_count": 0,935    "user_voted": false,936    "discourse_zendesk_plugin_zendesk_id": null,937    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",938    "details": {939      "can_edit": false,940      "notification_level": 1,941      "participants": [942        {943          "id": 58587,944          "username": "Omega_Ma",945          "name": "Omega Ma",946          "avatar_template": "/user_avatar/discuss.pytorch.org/omega_ma/{size}/52185_2.png",947          "post_count": 2,948          "primary_group_name": null,949          "flair_name": null,950          "flair_url": null,951          "flair_color": null,952          "flair_bg_color": null,953          "flair_group_id": null,954          "trust_level": 1955        },956        {957          "id": 61114,958          "username": "pentachris",959          "name": "Chris",960          "avatar_template": "/letter_avatar_proxy/v4/letter/p/dec6dc/{size}.png",961          "post_count": 1,962          "primary_group_name": null,963          "flair_name": null,964          "flair_url": null,965          "flair_color": null,966          "flair_bg_color": null,967          "flair_group_id": null,968          "trust_level": 1969        }970      ],971      "created_by": {972        "id": 58587,973        "username": "Omega_Ma",974        "name": "Omega Ma",975        "avatar_template": "/user_avatar/discuss.pytorch.org/omega_ma/{size}/52185_2.png"976      },977      "last_poster": {978        "id": 58587,979        "username": "Omega_Ma",980        "name": "Omega Ma",981        "avatar_template": "/user_avatar/discuss.pytorch.org/omega_ma/{size}/52185_2.png"982      }983    },984    "bookmarks": []985  },986  {987    "post_stream": {988      "posts": [989        {990          "id": 375469,991          "name": "",992          "username": "Tm4",993          "avatar_template": "/letter_avatar_proxy/v4/letter/t/b38774/{size}.png",994          "created_at": "2022-11-19T19:50:53.503Z",995          "cooked": "<p>In my program, I use three pretrained encoders and for training my model I am using DataParallel. I want to use three Gpus that each of them has different free capacity. I am usung DataParallel in the following scenario:</p>\n<pre><code>dev = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = nn.DataParallel(model, output_device=[0,1,2], device_ids = [0,1,2])\nmodel = model.to(dev)\n</code></pre>\n<p>Unfortunately, the program doesn’t distribute the required memory on the three device automatically and it wants to use beyond the capacity of one of the devices though other two devices have required free space. It shows the error that one of the devices doesn’t have enough memory but at the same time most of the memory space of other two gpus are free. How can I fix that?<br>\n<div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/3X/2/b/2be4e6665cc574cde03eb56294a79b603935012a.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/2be4e6665cc574cde03eb56294a79b603935012a\" title=\"image\"><img src=\"https://discuss.pytorch.org/uploads/default/original/3X/2/b/2be4e6665cc574cde03eb56294a79b603935012a.png\" alt=\"image\" data-base62-sha1=\"6giWJ6aonQw53CnfNL6VrVoG2Ey\" width=\"690\" height=\"40\" data-dominant-color=\"DBCBCA\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">image</span><span class=\"informations\">1253×73 7.44 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div></p>",996          "post_number": 1,997          "post_type": 1,998          "posts_count": 1,999          "updated_at": "2022-11-19T19:52:28.451Z",1000          "reply_count": 0,1001          "reply_to_post_number": null,1002          "quote_count": 0,1003          "incoming_link_count": 10,1004          "reads": 2,1005          "readers_count": 1,1006          "score": 50.4,1007          "yours": false,1008          "topic_id": 166412,1009          "topic_slug": "aloccating-required-memory-for-each-gpu-in-data-parallelism-automatically",1010          "display_username": "",1011          "primary_group_name": null,1012          "flair_name": null,1013          "flair_url": null,1014          "flair_bg_color": null,1015          "flair_color": null,1016          "flair_group_id": null,1017          "badges_granted": [],1018          "version": 1,1019          "can_edit": false,1020          "can_delete": false,1021          "can_recover": false,1022          "can_see_hidden_post": false,1023          "can_wiki": false,1024          "link_counts": [1025            {1026              "url": "https://discuss.pytorch.org/uploads/default/original/3X/2/b/2be4e6665cc574cde03eb56294a79b603935012a.png",1027              "internal": true,1028              "reflection": false,1029              "clicks": 01030            }1031          ],1032          "read": true,1033          "user_title": null,1034          "bookmarked": false,1035          "actions_summary": [],1036          "moderator": false,1037          "admin": false,1038          "staff": false,1039          "user_id": 55068,1040          "hidden": false,1041          "trust_level": 1,1042          "deleted_at": null,1043          "user_deleted": false,1044          "edit_reason": null,1045          "can_view_edit_history": true,1046          "wiki": false,1047          "post_url": "/t/aloccating-required-memory-for-each-gpu-in-data-parallelism-automatically/166412/1",1048          "can_accept_answer": false,1049          "can_unaccept_answer": false,1050          "accepted_answer": false,1051          "topic_accepted_answer": null,1052          "can_vote": false1053        }1054      ],1055      "stream": [1056        3754691057      ]1058    },1059    "timeline_lookup": [1060      [1061        1,1062        10711063      ]1064    ],1065    "suggested_topics": [1066      {1067        "fancy_title": "Looking to Contribute: Understanding PyTorch Architecture/Framework",1068        "id": 213595,1069        "title": "Looking to Contribute: Understanding PyTorch Architecture/Framework",1070        "slug": "looking-to-contribute-understanding-pytorch-architecture-framework",1071        "posts_count": 2,1072        "reply_count": 0,1073        "highest_post_number": 3,1074        "image_url": null,1075        "created_at": "2024-11-29T00:30:25.768Z",1076        "last_posted_at": "2024-12-15T17:21:21.085Z",1077        "bumped": true,1078        "bumped_at": "2024-12-15T17:21:21.085Z",1079        "archetype": "regular",1080        "unseen": false,1081        "pinned": false,1082        "unpinned": null,1083        "visible": true,1084        "closed": false,1085        "archived": false,1086        "bookmarked": null,1087        "liked": null,1088        "tags_descriptions": {},1089        "like_count": 0,1090        "views": 62,1091        "category_id": 1,1092        "featured_link": null,1093        "has_accepted_answer": false,1094        "posters": [1095          {1096            "extras": null,1097            "description": "Original Poster",1098            "user": {1099              "id": 81199,1100              "username": "Emmett_Bicker",1101              "name": "Emmett Bicker",1102              "avatar_template": "/user_avatar/discuss.pytorch.org/emmett_bicker/{size}/74263_2.png",1103              "trust_level": 01104            }1105          },1106          {1107            "extras": "latest",1108            "description": "Most Recent Poster",1109            "user": {1110              "id": 41396,1111              "username": "soulitzer",1112              "name": "",1113              "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",1114              "trust_level": 21115            }1116          }1117        ]1118      },1119      {1120        "fancy_title": "NVIDIA L40S-48Q and &ldquo;RuntimeError: CUDA error: operation not supported&rdquo;",1121        "id": 212716,1122        "title": "NVIDIA L40S-48Q and \"RuntimeError: CUDA error: operation not supported\"",1123        "slug": "nvidia-l40s-48q-and-runtimeerror-cuda-error-operation-not-supported",1124        "posts_count": 11,1125        "reply_count": 9,1126        "highest_post_number": 11,1127        "image_url": null,1128        "created_at": "2024-11-08T19:29:41.292Z",1129        "last_posted_at": "2025-02-11T14:25:17.459Z",1130        "bumped": true,1131        "bumped_at": "2025-02-11T14:26:35.274Z",1132        "archetype": "regular",1133        "unseen": false,1134        "pinned": false,1135        "unpinned": null,1136        "visible": true,1137        "closed": false,1138        "archived": false,1139        "bookmarked": null,1140        "liked": null,1141        "tags_descriptions": {},1142        "like_count": 1,1143        "views": 1360,1144        "category_id": 1,1145        "featured_link": null,1146        "has_accepted_answer": false,1147        "posters": [1148          {1149            "extras": null,1150            "description": "Original Poster",1151            "user": {1152              "id": 7291,1153              "username": "Chris_Palmer",1154              "name": "Chris Palmer",1155              "avatar_template": "/user_avatar/discuss.pytorch.org/chris_palmer/{size}/12322_2.png",1156              "trust_level": 11157            }1158          },1159          {1160            "extras": null,1161            "description": "Frequent Poster",1162            "user": {1163              "id": 3534,1164              "username": "ptrblck",1165              "name": "",1166              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1167              "admin": true,1168              "moderator": true,1169              "trust_level": 21170            }1171          },1172          {1173            "extras": "latest",1174            "description": "Most Recent Poster",1175            "user": {1176              "id": 82625,1177              "username": "briskajanis1",1178              "name": "Briskajanis1",1179              "avatar_template": "/user_avatar/discuss.pytorch.org/briskajanis1/{size}/75599_2.png",1180              "trust_level": 01181            }1182          }1183        ]1184      },1185      {1186        "fancy_title": "Permute inside nn.Sequential",1187        "id": 214877,1188        "title": "Permute inside nn.Sequential",1189        "slug": "permute-inside-nn-sequential",1190        "posts_count": 3,1191        "reply_count": 0,1192        "highest_post_number": 3,1193        "image_url": null,1194        "created_at": "2025-01-02T08:19:47.254Z",1195        "last_posted_at": "2025-09-15T19:19:55.247Z",1196        "bumped": true,1197        "bumped_at": "2025-09-15T19:19:55.247Z",1198        "archetype": "regular",1199        "unseen": false,1200        "pinned": false,

Showing the first 1,200 of 58526 lines. Download the file for the rest.