CoolFace
Datasetpublic

Anurag1734/cuda-error-resolution-analysis

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes7downloads
topics_batch_484.json62538 linesDownload Raw Back to raw
1[2  {3    "post_stream": {4      "posts": [5        {6          "id": 193970,7          "name": "Ali Amiri",8          "username": "Ali_Amiri",9          "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png",10          "created_at": "2020-05-18T02:30:14.709Z",11          "cooked": "<p>Hi,<br>\nI have trained a transformer (from scratch) in PyTorch (many times in many ways) but steel gives me a terrible accuracy just like it is still random<br>\nbut in Keras, I trained the same network once and it gave me absolutely perfect performance even with a 5 samples dataset (I tested the model with same 5 samples in both frameworks but still Keras gives great result )<br>\nI checked the architecture many times but I haven’t figure out the problem yet!<br>\nI reeeaally appreciate if you help to me and other people with the same problem</p>\n<p>I think it’s worth nothing to say that if I give whole input and output to the model (in inference phase) it would work perfectly (I mean without using teacher forcing it works)</p>\n<p>notebooks are here:<br>\nKeras: <a href=\"https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing\" rel=\"nofollow noopener\">https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing</a><br>\nPyTorch: <a href=\"https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing\" rel=\"nofollow noopener\">https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing</a></p>",12          "post_number": 1,13          "post_type": 1,14          "posts_count": 1,15          "updated_at": "2020-05-18T02:38:13.787Z",16          "reply_count": 0,17          "reply_to_post_number": null,18          "quote_count": 0,19          "incoming_link_count": 14,20          "reads": 6,21          "readers_count": 5,22          "score": 71.2,23          "yours": false,24          "topic_id": 81725,25          "topic_slug": "i-trained-same-network-with-pytorch-but-totally-bad-results",26          "display_username": "Ali Amiri",27          "primary_group_name": null,28          "flair_name": null,29          "flair_url": null,30          "flair_bg_color": null,31          "flair_color": null,32          "flair_group_id": null,33          "badges_granted": [],34          "version": 2,35          "can_edit": false,36          "can_delete": false,37          "can_recover": false,38          "can_see_hidden_post": false,39          "can_wiki": false,40          "link_counts": [41            {42              "url": "https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing",43              "internal": false,44              "reflection": false,45              "clicks": 146            },47            {48              "url": "https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing",49              "internal": false,50              "reflection": false,51              "clicks": 152            }53          ],54          "read": true,55          "user_title": null,56          "bookmarked": false,57          "actions_summary": [],58          "moderator": false,59          "admin": false,60          "staff": false,61          "user_id": 30085,62          "hidden": false,63          "trust_level": 1,64          "deleted_at": null,65          "user_deleted": false,66          "edit_reason": null,67          "can_view_edit_history": true,68          "wiki": false,69          "post_url": "/t/i-trained-same-network-with-pytorch-but-totally-bad-results/81725/1",70          "can_accept_answer": false,71          "can_unaccept_answer": false,72          "accepted_answer": false,73          "topic_accepted_answer": null,74          "can_vote": false75        }76      ],77      "stream": [78        19397079      ]80    },81    "timeline_lookup": [82      [83        1,84        198785      ]86    ],87    "suggested_topics": [88      {89        "fancy_title": "Transformer Stuck in Local Minima Occasionally",90        "id": 217970,91        "title": "Transformer Stuck in Local Minima Occasionally",92        "slug": "transformer-stuck-in-local-minima-occasionally",93        "posts_count": 1,94        "reply_count": 0,95        "highest_post_number": 1,96        "image_url": null,97        "created_at": "2025-03-18T03:33:44.966Z",98        "last_posted_at": "2025-03-18T03:33:44.998Z",99        "bumped": true,100        "bumped_at": "2025-03-18T03:33:44.998Z",101        "archetype": "regular",102        "unseen": false,103        "pinned": false,104        "unpinned": null,105        "visible": true,106        "closed": false,107        "archived": false,108        "bookmarked": null,109        "liked": null,110        "tags_descriptions": {},111        "like_count": 0,112        "views": 63,113        "category_id": 8,114        "featured_link": null,115        "has_accepted_answer": false,116        "posters": [117          {118            "extras": "latest single",119            "description": "Original Poster, Most Recent Poster",120            "user": {121              "id": 83338,122              "username": "martin12",123              "name": "Martin",124              "avatar_template": "/user_avatar/discuss.pytorch.org/martin12/{size}/76222_2.png",125              "trust_level": 0126            }127          }128        ]129      },130      {131        "fancy_title": "A Simple LSTM stuck into label flipping",132        "id": 220004,133        "title": "A Simple LSTM stuck into label flipping",134        "slug": "a-simple-lstm-stuck-into-label-flipping",135        "posts_count": 5,136        "reply_count": 3,137        "highest_post_number": 5,138        "image_url": null,139        "created_at": "2025-05-13T19:44:43.099Z",140        "last_posted_at": "2025-05-15T08:28:26.193Z",141        "bumped": true,142        "bumped_at": "2025-05-15T08:28:26.193Z",143        "archetype": "regular",144        "unseen": false,145        "pinned": false,146        "unpinned": null,147        "visible": true,148        "closed": false,149        "archived": false,150        "bookmarked": null,151        "liked": null,152        "tags_descriptions": {},153        "like_count": 0,154        "views": 72,155        "category_id": 8,156        "featured_link": null,157        "has_accepted_answer": true,158        "posters": [159          {160            "extras": "latest",161            "description": "Original Poster, Most Recent Poster",162            "user": {163              "id": 84265,164              "username": "subhojitdas",165              "name": "Subhojit Das",166              "avatar_template": "/user_avatar/discuss.pytorch.org/subhojitdas/{size}/77001_2.png",167              "trust_level": 1168            }169          },170          {171            "extras": null,172            "description": "Frequent Poster, Accepted Answer",173            "user": {174              "id": 1438,175              "username": "vdw",176              "name": "Chris",177              "avatar_template": "/user_avatar/discuss.pytorch.org/vdw/{size}/10074_2.png",178              "trust_level": 2179            }180          }181        ]182      },183      {184        "fancy_title": "LSTM for classification (fraud detection) over several lines of text",185        "id": 216350,186        "title": "LSTM for classification (fraud detection) over several lines of text",187        "slug": "lstm-for-classification-fraud-detection-over-several-lines-of-text",188        "posts_count": 1,189        "reply_count": 0,190        "highest_post_number": 1,191        "image_url": null,192        "created_at": "2025-02-07T10:42:24.333Z",193        "last_posted_at": "2025-02-07T10:42:24.380Z",194        "bumped": true,195        "bumped_at": "2025-02-07T11:24:44.201Z",196        "archetype": "regular",197        "unseen": false,198        "pinned": false,199        "unpinned": null,200        "visible": true,201        "closed": false,202        "archived": false,203        "bookmarked": null,204        "liked": null,205        "tags_descriptions": {},206        "like_count": 0,207        "views": 144,208        "category_id": 8,209        "featured_link": null,210        "has_accepted_answer": false,211        "posters": [212          {213            "extras": "latest single",214            "description": "Original Poster, Most Recent Poster",215            "user": {216              "id": 82539,217              "username": "Monkee_Motion",218              "name": "Monkee Motion",219              "avatar_template": "/user_avatar/discuss.pytorch.org/monkee_motion/{size}/75523_2.png",220              "trust_level": 0221            }222          }223        ]224      },225      {226        "fancy_title": "How to compute the Validation loss",227        "id": 213365,228        "title": "How to compute the Validation loss",229        "slug": "how-to-compute-the-validation-loss",230        "posts_count": 3,231        "reply_count": 1,232        "highest_post_number": 3,233        "image_url": null,234        "created_at": "2024-11-24T08:21:30.769Z",235        "last_posted_at": "2024-11-24T16:57:48.437Z",236        "bumped": true,237        "bumped_at": "2024-11-24T16:57:48.437Z",238        "archetype": "regular",239        "unseen": false,240        "pinned": false,241        "unpinned": null,242        "visible": true,243        "closed": false,244        "archived": false,245        "bookmarked": null,246        "liked": null,247        "tags_descriptions": {},248        "like_count": 0,249        "views": 61,250        "category_id": 8,251        "featured_link": null,252        "has_accepted_answer": false,253        "posters": [254          {255            "extras": "latest",256            "description": "Original Poster, Most Recent Poster",257            "user": {258              "id": 81094,259              "username": "ali_fahad",260              "name": "ali fahad",261              "avatar_template": "/user_avatar/discuss.pytorch.org/ali_fahad/{size}/74160_2.png",262              "trust_level": 0263            }264          },265          {266            "extras": null,267            "description": "Frequent Poster",268            "user": {269              "id": 3534,270              "username": "ptrblck",271              "name": "",272              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",273              "admin": true,274              "moderator": true,275              "trust_level": 2276            }277          }278        ]279      },280      {281        "fancy_title": "Gemma 3 throws RuntimeError CUDA misaligned address",282        "id": 220507,283        "title": "Gemma 3 throws RuntimeError CUDA misaligned address",284        "slug": "gemma-3-throws-runtimeerror-cuda-misaligned-address",285        "posts_count": 2,286        "reply_count": 0,287        "highest_post_number": 2,288        "image_url": null,289        "created_at": "2025-06-02T07:00:32.299Z",290        "last_posted_at": "2025-06-03T22:25:43.492Z",291        "bumped": true,292        "bumped_at": "2025-06-03T22:25:43.492Z",293        "archetype": "regular",294        "unseen": false,295        "pinned": false,296        "unpinned": null,297        "visible": true,298        "closed": false,299        "archived": false,300        "bookmarked": null,301        "liked": null,302        "tags_descriptions": {},303        "like_count": 0,304        "views": 116,305        "category_id": 8,306        "featured_link": null,307        "has_accepted_answer": false,308        "posters": [309          {310            "extras": null,311            "description": "Original Poster",312            "user": {313              "id": 84544,314              "username": "msi-sbraun-11",315              "name": "",316              "avatar_template": "/user_avatar/discuss.pytorch.org/msi-sbraun-11/{size}/77240_2.png",317              "trust_level": 0318            }319          },320          {321            "extras": "latest",322            "description": "Most Recent Poster",323            "user": {324              "id": 3534,325              "username": "ptrblck",326              "name": "",327              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",328              "admin": true,329              "moderator": true,330              "trust_level": 2331            }332          }333        ]334      }335    ],336    "tags_descriptions": {},337    "fancy_title": "I trained same network with pytorch but totally bad results",338    "id": 81725,339    "title": "I trained same network with pytorch but totally bad results",340    "posts_count": 1,341    "created_at": "2020-05-18T02:30:14.640Z",342    "views": 357,343    "reply_count": 0,344    "like_count": 0,345    "last_posted_at": "2020-05-18T02:30:14.709Z",346    "visible": true,347    "closed": false,348    "archived": false,349    "has_summary": false,350    "archetype": "regular",351    "slug": "i-trained-same-network-with-pytorch-but-totally-bad-results",352    "category_id": 8,353    "word_count": 155,354    "deleted_at": null,355    "user_id": 30085,356    "featured_link": null,357    "pinned_globally": false,358    "pinned_at": null,359    "pinned_until": null,360    "image_url": null,361    "slow_mode_seconds": 0,362    "draft": null,363    "draft_key": "topic_81725",364    "draft_sequence": null,365    "unpinned": null,366    "pinned": false,367    "current_post_number": 1,368    "highest_post_number": 1,369    "deleted_by": null,370    "actions_summary": [371      {372        "id": 4,373        "count": 0,374        "hidden": false,375        "can_act": false376      },377      {378        "id": 8,379        "count": 0,380        "hidden": false,381        "can_act": false382      },383      {384        "id": 10,385        "count": 0,386        "hidden": false,387        "can_act": false388      },389      {390        "id": 7,391        "count": 0,392        "hidden": false,393        "can_act": false394      }395    ],396    "chunk_size": 20,397    "bookmarked": false,398    "topic_timer": null,399    "message_bus_last_id": 0,400    "participant_count": 1,401    "show_read_indicator": false,402    "thumbnails": null,403    "slow_mode_enabled_until": null,404    "can_vote": false,405    "vote_count": 0,406    "user_voted": false,407    "discourse_zendesk_plugin_zendesk_id": null,408    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",409    "details": {410      "can_edit": false,411      "notification_level": 1,412      "participants": [413        {414          "id": 30085,415          "username": "Ali_Amiri",416          "name": "Ali Amiri",417          "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png",418          "post_count": 1,419          "primary_group_name": null,420          "flair_name": null,421          "flair_url": null,422          "flair_color": null,423          "flair_bg_color": null,424          "flair_group_id": null,425          "trust_level": 1426        }427      ],428      "created_by": {429        "id": 30085,430        "username": "Ali_Amiri",431        "name": "Ali Amiri",432        "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png"433      },434      "last_poster": {435        "id": 30085,436        "username": "Ali_Amiri",437        "name": "Ali Amiri",438        "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png"439      },440      "links": [441        {442          "url": "https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing",443          "title": null,444          "internal": false,445          "attachment": false,446          "reflection": false,447          "clicks": 1,448          "user_id": 30085,449          "domain": "colab.research.google.com",450          "root_domain": "google.com"451        },452        {453          "url": "https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing",454          "title": null,455          "internal": false,456          "attachment": false,457          "reflection": false,458          "clicks": 1,459          "user_id": 30085,460          "domain": "colab.research.google.com",461          "root_domain": "google.com"462        }463      ]464    },465    "bookmarks": []466  },467  {468    "post_stream": {469      "posts": [470        {471          "id": 193950,472          "name": "Shantanu Ghosh",473          "username": "Shantanu_Ghosh",474          "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png",475          "created_at": "2020-05-18T01:15:42.733Z",476          "cooked": "<p>Hi,<br>\nI have trained an autoencoder to get the lower dimensional representation of a image dataset.<br>\nNow I have saved the model. Then I load it and I want to remove the weights of the decoders. With the weights of the encoder I want to intiatlise the conv layers of a CNN classifier. This CNN classifier has same structure with the encoder of the autoencoder. Can anyone help me?</p>\n<p>Basically I want to implement this project(implemented in keras)in pytorch:<br>\n</p><aside class=\"onebox whitelistedgeneric\">\n  <header class=\"source\">\n      <img src=\"https://cdn.datacamp.com/main-app/assets/favicon-335cd0394b32102a39221d79e5fd7e51078e6d32a0c8aea59676a6869f84e9d8.ico\" class=\"site-icon\" width=\"64\" height=\"64\">\n      <a href=\"https://www.datacamp.com/community/tutorials/autoencoder-classifier-python\" target=\"_blank\" rel=\"nofollow noopener\" title=\"04:00PM - 20 July 2018\">DataCamp Community – 20 Jul 18</a>\n  </header>\n  <article class=\"onebox-body\">\n    <img src=\"https://cdn.datacamp.com/community/logos/medium/placeholder.jpg\" class=\"thumbnail onebox-avatar\" width=\"240\" height=\"240\">\n\n<h3><a href=\"https://www.datacamp.com/community/tutorials/autoencoder-classifier-python\" target=\"_blank\" rel=\"nofollow noopener\">Autoencoder as a Classifier</a></h3>\n\n<p>In this tutorial, you will learn &amp; understand how to use autoencoder as a classifier in Python with Keras. You'll be using Fashion-MNIST dataset as an example.</p>\n\n\n  </article>\n  <div class=\"onebox-metadata\">\n    \n    \n  </div>\n  <div style=\"clear: both\"></div>\n</aside>\n",477          "post_number": 1,478          "post_type": 1,479          "posts_count": 2,480          "updated_at": "2020-05-18T02:57:06.123Z",481          "reply_count": 1,482          "reply_to_post_number": null,483          "quote_count": 0,484          "incoming_link_count": 150,485          "reads": 12,486          "readers_count": 11,487          "score": 757.4,488          "yours": false,489          "topic_id": 81713,490          "topic_slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",491          "display_username": "Shantanu Ghosh",492          "primary_group_name": null,493          "flair_name": null,494          "flair_url": null,495          "flair_bg_color": null,496          "flair_color": null,497          "flair_group_id": null,498          "badges_granted": [],499          "version": 3,500          "can_edit": false,501          "can_delete": false,502          "can_recover": false,503          "can_see_hidden_post": false,504          "can_wiki": false,505          "link_counts": [506            {507              "url": "https://www.datacamp.com/community/tutorials/autoencoder-classifier-python",508              "internal": false,509              "reflection": false,510              "title": "Autoencoder as a Classifier - DataCamp",511              "clicks": 10512            }513          ],514          "read": true,515          "user_title": null,516          "bookmarked": false,517          "actions_summary": [],518          "moderator": false,519          "admin": false,520          "staff": false,521          "user_id": 31776,522          "hidden": false,523          "trust_level": 1,524          "deleted_at": null,525          "user_deleted": false,526          "edit_reason": null,527          "can_view_edit_history": true,528          "wiki": false,529          "post_url": "/t/initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder/81713/1",530          "can_accept_answer": false,531          "can_unaccept_answer": false,532          "accepted_answer": false,533          "topic_accepted_answer": null,534          "can_vote": false535        },536        {537          "id": 193967,538          "name": "Jaya Krishna Mandivarapu",539          "username": "jmandivarapu1",540          "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png",541          "created_at": "2020-05-18T02:28:21.761Z",542          "cooked": "<aside class=\"quote no-group quote-modified\" data-username=\"Shantanu_Ghosh\" data-post=\"1\" data-topic=\"81713\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/user_avatar/discuss.pytorch.org/shantanu_ghosh/48/24369_2.png\" class=\"avatar\"> Shantanu_Ghosh:</div>\n<blockquote>\n<p>class CNN(nn.Module):<br>\ndef <strong>init</strong> (self):<br>\nsuper(CNN, self). <strong>init</strong> ()</p>\n<pre><code class=\"lang-auto\">    # Conv layers\n    self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n    self.bn1 = nn.BatchNorm2d(96)\n    self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n    self.bn2 = nn.BatchNorm2d(256)\n    self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n    self.pool = nn.MaxPool2d(3, 2)\n    # FC layers\n</code></pre>\n</blockquote>\n</aside>\n<pre><code class=\"lang-auto\">class Autoencoder(nn.Module):\n    def __init__(self):\n        super(Autoencoder, self).__init__()\n          # encoder layers\n        self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n        self.bn1 = nn.BatchNorm2d(96)\n        self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n        self.bn2 = nn.BatchNorm2d(256)\n        self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n        self.pool = nn.MaxPool2d(3, 2)\n\n        # decoder layers\n        self.dec1 = nn.ConvTranspose2d(384, 256, kernel_size=3, stride=2)\n        self.dec2 = nn.ConvTranspose2d(256, 96, kernel_size=3, stride=2)\n        self.dec3 = nn.ConvTranspose2d(96, 3, kernel_size=11, stride=4)\n\n        def forward(self, x):\n\n            # encoder\n            x = F.relu(self.bn1(self.enc1(x)))\n            x = self.pool(x)\n            x = F.relu(self.bn2(self.enc2(x)))\n            x = self.pool(x)\n            x = F.relu(self.enc3(x))  # the latent space representation\n\n            # decoder\n            x = F.relu(self.dec1(x))\n            x = F.relu(self.dec2(x))\n            x = F.relu(self.dec3(x))\n            # x = F.relu(self.dec4(x))\n            # x = F.sigmoid(self.out(x))\n\n            return x\nclass CNN(nn.Module):\n    def __init__(self):\n        super(CNN, self).__init__()\n\n        # Conv layers\n        self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n        self.bn1 = nn.BatchNorm2d(96)\n        self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n        self.bn2 = nn.BatchNorm2d(256)\n        self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n        self.pool = nn.MaxPool2d(3, 2)\n        # FC layers\n        \nx=Autoencoder()\ny=CNN()\nprint(x.enc1.weight)\nx.enc1.weight=y.enc1.weight\n\nprint(x.enc1.weight)\n</code></pre>",543          "post_number": 2,544          "post_type": 1,545          "posts_count": 2,546          "updated_at": "2020-05-18T02:28:21.761Z",547          "reply_count": 0,548          "reply_to_post_number": null,549          "quote_count": 1,550          "incoming_link_count": 10,551          "reads": 12,552          "readers_count": 11,553          "score": 52.4,554          "yours": false,555          "topic_id": 81713,556          "topic_slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",557          "display_username": "Jaya Krishna Mandivarapu",558          "primary_group_name": null,559          "flair_name": null,560          "flair_url": null,561          "flair_bg_color": null,562          "flair_color": null,563          "flair_group_id": null,564          "badges_granted": [],565          "version": 1,566          "can_edit": false,567          "can_delete": false,568          "can_recover": false,569          "can_see_hidden_post": false,570          "can_wiki": false,571          "read": true,572          "user_title": "",573          "bookmarked": false,574          "actions_summary": [],575          "moderator": false,576          "admin": false,577          "staff": false,578          "user_id": 3781,579          "hidden": false,580          "trust_level": 2,581          "deleted_at": null,582          "user_deleted": false,583          "edit_reason": null,584          "can_view_edit_history": true,585          "wiki": false,586          "post_url": "/t/initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder/81713/2",587          "can_accept_answer": false,588          "can_unaccept_answer": false,589          "accepted_answer": false,590          "topic_accepted_answer": null591        }592      ],593      "stream": [594        193950,595        193967596      ]597    },598    "timeline_lookup": [599      [600        1,601        1987602      ]603    ],604    "suggested_topics": [605      {606        "fancy_title": "Batch size at inference is influencing accuracy",607        "id": 218096,608        "title": "Batch size at inference is influencing accuracy",609        "slug": "batch-size-at-inference-is-influencing-accuracy",610        "posts_count": 1,611        "reply_count": 0,612        "highest_post_number": 1,613        "image_url": null,614        "created_at": "2025-03-20T21:02:58.000Z",615        "last_posted_at": "2025-03-20T21:02:58.052Z",616        "bumped": true,617        "bumped_at": "2025-03-20T21:02:58.052Z",618        "archetype": "regular",619        "unseen": false,620        "pinned": false,621        "unpinned": null,622        "visible": true,623        "closed": false,624        "archived": false,625        "bookmarked": null,626        "liked": null,627        "tags_descriptions": {},628        "like_count": 0,629        "views": 45,630        "category_id": 5,631        "featured_link": null,632        "has_accepted_answer": false,633        "posters": [634          {635            "extras": "latest single",636            "description": "Original Poster, Most Recent Poster",637            "user": {638              "id": 83393,639              "username": "danbull-scanabull",640              "name": "Danbull Scanabull",641              "avatar_template": "/user_avatar/discuss.pytorch.org/danbull-scanabull/{size}/76268_2.png",642              "trust_level": 1643            }644          }645        ]646      },647      {648        "fancy_title": "TypeError: &lsquo;int&rsquo; object is not callable for claculating training accuracy",649        "id": 212609,650        "title": "TypeError: 'int' object is not callable for claculating training accuracy",651        "slug": "typeerror-int-object-is-not-callable-for-claculating-training-accuracy",652        "posts_count": 2,653        "reply_count": 0,654        "highest_post_number": 2,655        "image_url": null,656        "created_at": "2024-11-06T11:04:54.598Z",657        "last_posted_at": "2024-11-06T17:20:42.372Z",658        "bumped": true,659        "bumped_at": "2024-11-06T17:20:42.372Z",660        "archetype": "regular",661        "unseen": false,662        "pinned": false,663        "unpinned": null,664        "visible": true,665        "closed": false,666        "archived": false,667        "bookmarked": null,668        "liked": null,669        "tags_descriptions": {},670        "like_count": 0,671        "views": 30,672        "category_id": 5,673        "featured_link": null,674        "has_accepted_answer": false,675        "posters": [676          {677            "extras": null,678            "description": "Original Poster",679            "user": {680              "id": 68222,681              "username": "amy2",682              "name": "amy",683              "avatar_template": "/user_avatar/discuss.pytorch.org/amy2/{size}/62675_2.png",684              "trust_level": 1685            }686          },687          {688            "extras": "latest",689            "description": "Most Recent Poster",690            "user": {691              "id": 3534,692              "username": "ptrblck",693              "name": "",694              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",695              "admin": true,696              "moderator": true,697              "trust_level": 2698            }699          }700        ]701      },702      {703        "fancy_title": "Extracting Swin-Vit backbone",704        "id": 212799,705        "title": "Extracting Swin-Vit backbone",706        "slug": "extracting-swin-vit-backbone",707        "posts_count": 1,708        "reply_count": 0,709        "highest_post_number": 1,710        "image_url": null,711        "created_at": "2024-11-11T09:54:48.491Z",712        "last_posted_at": "2024-11-11T09:54:48.540Z",713        "bumped": true,714        "bumped_at": "2024-11-11T09:54:48.540Z",715        "archetype": "regular",716        "unseen": false,717        "pinned": false,718        "unpinned": null,719        "visible": true,720        "closed": false,721        "archived": false,722        "bookmarked": null,723        "liked": null,724        "tags_descriptions": {},725        "like_count": 0,726        "views": 196,727        "category_id": 5,728        "featured_link": null,729        "has_accepted_answer": false,730        "posters": [731          {732            "extras": "latest single",733            "description": "Original Poster, Most Recent Poster",734            "user": {735              "id": 56688,736              "username": "ima",737              "name": "Imantha gunasekera",738              "avatar_template": "/letter_avatar_proxy/v4/letter/i/67e7ee/{size}.png",739              "trust_level": 1740            }741          }742        ]743      },744      {745        "fancy_title": "Semantic Segmentation with Attention based CycleGAN",746        "id": 217647,747        "title": "Semantic Segmentation with Attention based CycleGAN",748        "slug": "semantic-segmentation-with-attention-based-cyclegan",749        "posts_count": 5,750        "reply_count": 1,751        "highest_post_number": 5,752        "image_url": null,753        "created_at": "2025-03-10T06:44:59.887Z",754        "last_posted_at": "2025-03-11T22:57:33.781Z",755        "bumped": true,756        "bumped_at": "2025-03-11T22:57:33.781Z",757        "archetype": "regular",758        "unseen": false,759        "pinned": false,760        "unpinned": null,761        "visible": true,762        "closed": false,763        "archived": false,764        "bookmarked": null,765        "liked": null,766        "tags_descriptions": {},767        "like_count": 0,768        "views": 133,769        "category_id": 5,770        "featured_link": null,771        "has_accepted_answer": false,772        "posters": [773          {774            "extras": null,775            "description": "Original Poster",776            "user": {777              "id": 81422,778              "username": "Idrees11",779              "name": "Idrees Bhat",780              "avatar_template": "/user_avatar/discuss.pytorch.org/idrees11/{size}/74448_2.png",781              "trust_level": 1782            }783          },784          {785            "extras": "latest",786            "description": "Most Recent Poster",787            "user": {788              "id": 18088,789              "username": "KFrank",790              "name": "K. Frank",791              "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",792              "trust_level": 2793            }794          }795        ]796      },797      {798        "fancy_title": "My Loss function becomes 0 in 2nd epoch",799        "id": 221317,800        "title": "My Loss function becomes 0 in 2nd epoch",801        "slug": "my-loss-function-becomes-0-in-2nd-epoch",802        "posts_count": 2,803        "reply_count": 0,804        "highest_post_number": 2,805        "image_url": null,806        "created_at": "2025-07-07T05:55:40.826Z",807        "last_posted_at": "2025-07-08T22:59:30.226Z",808        "bumped": true,809        "bumped_at": "2025-07-08T22:59:30.226Z",810        "archetype": "regular",811        "unseen": false,812        "pinned": false,813        "unpinned": null,814        "visible": true,815        "closed": false,816        "archived": false,817        "bookmarked": null,818        "liked": null,819        "tags_descriptions": {},820        "like_count": 1,821        "views": 74,822        "category_id": 5,823        "featured_link": null,824        "has_accepted_answer": false,825        "posters": [826          {827            "extras": null,828            "description": "Original Poster",829            "user": {830              "id": 84964,831              "username": "syeda_raheen",832              "name": "Raheen",833              "avatar_template": "/letter_avatar_proxy/v4/letter/s/b19c9b/{size}.png",834              "trust_level": 0835            }836          },837          {838            "extras": "latest",839            "description": "Most Recent Poster",840            "user": {841              "id": 84484,842              "username": "Dhia-naouali",843              "name": "Dhia naouali",844              "avatar_template": "/user_avatar/discuss.pytorch.org/dhia-naouali/{size}/77193_2.png",845              "trust_level": 2846            }847          }848        ]849      }850    ],851    "tags_descriptions": {},852    "fancy_title": "Initialise the weights of the conv layers with the weights of a pretrained autoencoder",853    "id": 81713,854    "title": "Initialise the weights of the conv layers with the weights of a pretrained autoencoder",855    "posts_count": 2,856    "created_at": "2020-05-18T01:15:42.644Z",857    "views": 545,858    "reply_count": 0,859    "like_count": 0,860    "last_posted_at": "2020-05-18T02:28:21.761Z",861    "visible": true,862    "closed": false,863    "archived": false,864    "has_summary": false,865    "archetype": "regular",866    "slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",867    "category_id": 5,868    "word_count": 412,869    "deleted_at": null,870    "user_id": 31776,871    "featured_link": null,872    "pinned_globally": false,873    "pinned_at": null,874    "pinned_until": null,875    "image_url": null,876    "slow_mode_seconds": 0,877    "draft": null,878    "draft_key": "topic_81713",879    "draft_sequence": null,880    "unpinned": null,881    "pinned": false,882    "current_post_number": 1,883    "highest_post_number": 2,884    "deleted_by": null,885    "actions_summary": [886      {887        "id": 4,888        "count": 0,889        "hidden": false,890        "can_act": false891      },892      {893        "id": 8,894        "count": 0,895        "hidden": false,896        "can_act": false897      },898      {899        "id": 10,900        "count": 0,901        "hidden": false,902        "can_act": false903      },904      {905        "id": 7,906        "count": 0,907        "hidden": false,908        "can_act": false909      }910    ],911    "chunk_size": 20,912    "bookmarked": false,913    "topic_timer": null,914    "message_bus_last_id": 0,915    "participant_count": 2,916    "show_read_indicator": false,917    "thumbnails": null,918    "slow_mode_enabled_until": null,919    "can_vote": false,920    "vote_count": 0,921    "user_voted": false,922    "discourse_zendesk_plugin_zendesk_id": null,923    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",924    "details": {925      "can_edit": false,926      "notification_level": 1,927      "participants": [928        {929          "id": 3781,930          "username": "jmandivarapu1",931          "name": "Jaya Krishna Mandivarapu",932          "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png",933          "post_count": 1,934          "primary_group_name": null,935          "flair_name": null,936          "flair_url": null,937          "flair_color": null,938          "flair_bg_color": null,939          "flair_group_id": null,940          "trust_level": 2941        },942        {943          "id": 31776,944          "username": "Shantanu_Ghosh",945          "name": "Shantanu Ghosh",946          "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png",947          "post_count": 1,948          "primary_group_name": null,949          "flair_name": null,950          "flair_url": null,951          "flair_color": null,952          "flair_bg_color": null,953          "flair_group_id": null,954          "trust_level": 1955        }956      ],957      "created_by": {958        "id": 31776,959        "username": "Shantanu_Ghosh",960        "name": "Shantanu Ghosh",961        "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png"962      },963      "last_poster": {964        "id": 3781,965        "username": "jmandivarapu1",966        "name": "Jaya Krishna Mandivarapu",967        "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png"968      },969      "links": [970        {971          "url": "https://www.datacamp.com/community/tutorials/autoencoder-classifier-python",972          "title": "Autoencoder as a Classifier - DataCamp",973          "internal": false,974          "attachment": false,975          "reflection": false,976          "clicks": 10,977          "user_id": 31776,978          "domain": "www.datacamp.com",979          "root_domain": "datacamp.com"980        }981      ]982    },983    "bookmarks": []984  },985  {986    "post_stream": {987      "posts": [988        {989          "id": 192846,990          "name": "Zach",991          "username": "yangkz",992          "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",993          "created_at": "2020-05-14T18:10:46.029Z",994          "cooked": "<p>When I use two gpus to train my model, I got RuntimeError below:</p>\n<p>Process SpawnProcess-2:<br>\nTraceback (most recent call last):<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 258, in _bootstrap<br>\nself.run()<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 93, in run<br>\nself._target(*self._args, **self._kwargs)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 190, in run<br>\nmain(rank, dev_id, args)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 149, in main<br>\ntrain(args[‘gnn’], model, device, train_loader, criterion, optimizer, args[‘num_devices’], rank)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 41, in train<br>\noptimizer.backward_and_step(loss)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 146, in backward_and_step<br>\nself._sync_gradient()<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 127, in _sync_gradient<br>\ndist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/distributed_c10d.py”, line 902, in all_reduce<br>\nwork = _default_pg.allreduce([tensor], opts)<br>\nRuntimeError: Stop_waiting response is expected</p>\n<p>Process SpawnProcess-1:<br>\nTraceback (most recent call last):<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 258, in _bootstrap<br>\nself.run()<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 93, in run<br>\nself._target(*self._args, **self._kwargs)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 190, in run<br>\nmain(rank, dev_id, args)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 149, in main<br>\ntrain(args[‘gnn’], model, device, train_loader, criterion, optimizer, args[‘num_devices’], rank)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 41, in train<br>\noptimizer.backward_and_step(loss)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 146, in backward_and_step<br>\nself._sync_gradient()<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 127, in _sync_gradient<br>\ndist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/distributed_c10d.py”, line 902, in all_reduce<br>\nwork = _default_pg.allreduce([tensor], opts)<br>\nRuntimeError: Stop_waiting response is expected</p>\n<p>Here is the code where the error occurred:</p>\n<pre><code>def _sync_gradient(self):\n    \"\"\"Average gradients across all subprocesses.\"\"\"\n    for param_group in self.optimizer.param_groups:\n        for p in param_group['params']:\n            if p.requires_grad and p.grad is not None:\n                # print(p.grad.data.shape, p.grad.data.device)\n                dist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)\n                p.grad.data /= self.n_processes\n</code></pre>\n<p>Ps. When I do “print(p.grad.data.shape, p.grad.data.device)”, I find the grads are normal and has the same shape [1,300] on 2 different gpus. So I’m confused why it stopped here.</p>",995          "post_number": 1,996          "post_type": 1,997          "posts_count": 4,998          "updated_at": "2020-05-14T18:10:46.029Z",999          "reply_count": 0,1000          "reply_to_post_number": null,1001          "quote_count": 0,1002          "incoming_link_count": 1902,1003          "reads": 28,1004          "readers_count": 27,1005          "score": 9515.6,1006          "yours": false,1007          "topic_id": 81248,1008          "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1009          "display_username": "Zach",1010          "primary_group_name": null,1011          "flair_name": null,1012          "flair_url": null,1013          "flair_bg_color": null,1014          "flair_color": null,1015          "flair_group_id": null,1016          "badges_granted": [],1017          "version": 1,1018          "can_edit": false,1019          "can_delete": false,1020          "can_recover": false,1021          "can_see_hidden_post": false,1022          "can_wiki": false,1023          "read": true,1024          "user_title": "",1025          "bookmarked": false,1026          "actions_summary": [],1027          "moderator": false,1028          "admin": false,1029          "staff": false,1030          "user_id": 31619,1031          "hidden": false,1032          "trust_level": 1,1033          "deleted_at": null,1034          "user_deleted": false,1035          "edit_reason": null,1036          "can_view_edit_history": true,1037          "wiki": false,1038          "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/1",1039          "can_accept_answer": false,1040          "can_unaccept_answer": false,1041          "accepted_answer": false,1042          "topic_accepted_answer": true,1043          "can_vote": false1044        },1045        {1046          "id": 192854,1047          "name": "Shen Li",1048          "username": "mrshenli",1049          "avatar_template": "/user_avatar/discuss.pytorch.org/mrshenli/{size}/12220_2.png",1050          "created_at": "2020-05-14T19:07:41.991Z",1051          "cooked": "<p>Is the result of <code>p.requires_grad and p.grad is not None</code> always the same across all process and all parameters? If allreduce ops on different processes could run into desync.</p>\n<p>Which backend are you using (NCCL/Gloo/MPI) and which PyTorch version are you using? It will be helpful to have a min repro of this error.</p>",1052          "post_number": 2,1053          "post_type": 1,1054          "posts_count": 4,1055          "updated_at": "2020-05-14T19:07:41.991Z",1056          "reply_count": 1,1057          "reply_to_post_number": null,1058          "quote_count": 0,1059          "incoming_link_count": 6,1060          "reads": 28,1061          "readers_count": 27,1062          "score": 55.6,1063          "yours": false,1064          "topic_id": 81248,1065          "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1066          "display_username": "Shen Li",1067          "primary_group_name": null,1068          "flair_name": null,1069          "flair_url": null,1070          "flair_bg_color": null,1071          "flair_color": null,1072          "flair_group_id": null,1073          "badges_granted": [],1074          "version": 1,1075          "can_edit": false,1076          "can_delete": false,1077          "can_recover": false,1078          "can_see_hidden_post": false,1079          "can_wiki": false,1080          "read": true,1081          "user_title": null,1082          "bookmarked": false,1083          "actions_summary": [1084            {1085              "id": 2,1086              "count": 11087            }1088          ],1089          "moderator": false,1090          "admin": false,1091          "staff": false,1092          "user_id": 17068,1093          "hidden": false,1094          "trust_level": 2,1095          "deleted_at": null,1096          "user_deleted": false,1097          "edit_reason": null,1098          "can_view_edit_history": true,1099          "wiki": false,1100          "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/2",1101          "can_accept_answer": false,1102          "can_unaccept_answer": false,1103          "accepted_answer": false,1104          "topic_accepted_answer": true1105        },1106        {1107          "id": 193074,1108          "name": "Zach",1109          "username": "yangkz",1110          "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",1111          "created_at": "2020-05-15T07:55:04.015Z",1112          "cooked": "<p>Thanks for your reply!<br>\nHere is my short code:</p>\n<pre><code class=\"lang-auto\">def main(rank, dev_id, args):\n    torch.distributed.init_process_group(backend=\"nccl\",\n                                         init_method='tcp://localhost:22',\n                                         world_size=args['num_devices'],\n                                         rank=dev_id)\n    model = mymodel.to(dev_id)\n    optimizer = optim.Adam(model.parameters(), lr=args['lr'])\n    for epochs:\n        pred = model(inputs)\n        loss = criterion(pred, label)\n        optimizer.zero_grad()\n        loss.backward()\n\n        for param_group in optimizer.param_groups:\n            for p in param_group['params']:\n                if p.requires_grad and p.grad is not None:\n                    # print(p.grad.data.shape, p.grad.data.device) Ps. We can get grad information here\n                    dist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)\n                    p.grad.data /= n_processes\n        optimizer.step()\n    torch.distributed.barrier()\n\nmp = torch.multiprocessing.get_context('spawn')\nfor id, device_id in enumerate(devices):\n        procs.append(mp.Process(target=main, args=(id, device_id, args), daemon=True))\n        procs[-1].start()\nfor p in procs:\n    p.join()\n</code></pre>\n<p>The pytorch version I’m using is 1.4.0 and all my codes are running in AWS instance.<br>\nPs. The port of init_method can only be 22 which is strange otherwise it got Runtime Error like this:</p>\n<p>File “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/rendezvous.py”, line 120, in _tcp_rendezvous_handler<br>\nstore = TCPStore(result.hostname, result.port, world_size, start_daemon)<br>\nRuntimeError: connect() timed out.</p>\n<p>Thanks for your time and reading!</p>",1113          "post_number": 4,1114          "post_type": 1,1115          "posts_count": 4,1116          "updated_at": "2020-05-15T08:02:48.256Z",1117          "reply_count": 1,1118          "reply_to_post_number": 2,1119          "quote_count": 0,1120          "incoming_link_count": 160,1121          "reads": 27,1122          "readers_count": 26,1123          "score": 810.4,1124          "yours": false,1125          "topic_id": 81248,1126          "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1127          "display_username": "Zach",1128          "primary_group_name": null,1129          "flair_name": null,1130          "flair_url": null,1131          "flair_bg_color": null,1132          "flair_color": null,1133          "flair_group_id": null,1134          "badges_granted": [],1135          "version": 2,1136          "can_edit": false,1137          "can_delete": false,1138          "can_recover": false,1139          "can_see_hidden_post": false,1140          "can_wiki": false,1141          "read": true,1142          "user_title": "",1143          "reply_to_user": {1144            "id": 17068,1145            "username": "mrshenli",1146            "name": "Shen Li",1147            "avatar_template": "/user_avatar/discuss.pytorch.org/mrshenli/{size}/12220_2.png"1148          },1149          "bookmarked": false,1150          "actions_summary": [],1151          "moderator": false,1152          "admin": false,1153          "staff": false,1154          "user_id": 31619,1155          "hidden": false,1156          "trust_level": 1,1157          "deleted_at": null,1158          "user_deleted": false,1159          "edit_reason": null,1160          "can_view_edit_history": true,1161          "wiki": false,1162          "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/4",1163          "can_accept_answer": false,1164          "can_unaccept_answer": false,1165          "accepted_answer": false,1166          "topic_accepted_answer": true1167        },1168        {1169          "id": 193958,1170          "name": "Zach",1171          "username": "yangkz",1172          "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",1173          "created_at": "2020-05-18T01:46:31.076Z",1174          "cooked": "<p>The error has been fixed.<br>\n‘Stop_waiting response is expected’ error occurred in TCPStore.cpp. So it was actually the communication problem. It works finally when I reinstalled NCCL: <a href=\"https://github.com/NVIDIA/nccl.git\" rel=\"nofollow noopener\">https://github.com/NVIDIA/nccl.git</a></p>",1175          "post_number": 5,1176          "post_type": 1,1177          "posts_count": 4,1178          "updated_at": "2020-05-18T01:46:35.425Z",1179          "reply_count": 0,1180          "reply_to_post_number": 4,1181          "quote_count": 0,1182          "incoming_link_count": 21,1183          "reads": 26,1184          "readers_count": 25,1185          "score": 110.2,1186          "yours": false,1187          "topic_id": 81248,1188          "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1189          "display_username": "Zach",1190          "primary_group_name": null,1191          "flair_name": null,1192          "flair_url": null,1193          "flair_bg_color": null,1194          "flair_color": null,1195          "flair_group_id": null,1196          "badges_granted": [],1197          "version": 1,1198          "can_edit": false,1199          "can_delete": false,1200          "can_recover": false,

Showing the first 1,200 of 62538 lines. Download the file for the rest.