Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 193970,7 "name": "Ali Amiri",8 "username": "Ali_Amiri",9 "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png",10 "created_at": "2020-05-18T02:30:14.709Z",11 "cooked": "<p>Hi,<br>\nI have trained a transformer (from scratch) in PyTorch (many times in many ways) but steel gives me a terrible accuracy just like it is still random<br>\nbut in Keras, I trained the same network once and it gave me absolutely perfect performance even with a 5 samples dataset (I tested the model with same 5 samples in both frameworks but still Keras gives great result )<br>\nI checked the architecture many times but I haven’t figure out the problem yet!<br>\nI reeeaally appreciate if you help to me and other people with the same problem</p>\n<p>I think it’s worth nothing to say that if I give whole input and output to the model (in inference phase) it would work perfectly (I mean without using teacher forcing it works)</p>\n<p>notebooks are here:<br>\nKeras: <a href=\"https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing\" rel=\"nofollow noopener\">https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing</a><br>\nPyTorch: <a href=\"https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing\" rel=\"nofollow noopener\">https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing</a></p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 1,15 "updated_at": "2020-05-18T02:38:13.787Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 14,20 "reads": 6,21 "readers_count": 5,22 "score": 71.2,23 "yours": false,24 "topic_id": 81725,25 "topic_slug": "i-trained-same-network-with-pytorch-but-totally-bad-results",26 "display_username": "Ali Amiri",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 2,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "link_counts": [41 {42 "url": "https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing",43 "internal": false,44 "reflection": false,45 "clicks": 146 },47 {48 "url": "https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing",49 "internal": false,50 "reflection": false,51 "clicks": 152 }53 ],54 "read": true,55 "user_title": null,56 "bookmarked": false,57 "actions_summary": [],58 "moderator": false,59 "admin": false,60 "staff": false,61 "user_id": 30085,62 "hidden": false,63 "trust_level": 1,64 "deleted_at": null,65 "user_deleted": false,66 "edit_reason": null,67 "can_view_edit_history": true,68 "wiki": false,69 "post_url": "/t/i-trained-same-network-with-pytorch-but-totally-bad-results/81725/1",70 "can_accept_answer": false,71 "can_unaccept_answer": false,72 "accepted_answer": false,73 "topic_accepted_answer": null,74 "can_vote": false75 }76 ],77 "stream": [78 19397079 ]80 },81 "timeline_lookup": [82 [83 1,84 198785 ]86 ],87 "suggested_topics": [88 {89 "fancy_title": "Transformer Stuck in Local Minima Occasionally",90 "id": 217970,91 "title": "Transformer Stuck in Local Minima Occasionally",92 "slug": "transformer-stuck-in-local-minima-occasionally",93 "posts_count": 1,94 "reply_count": 0,95 "highest_post_number": 1,96 "image_url": null,97 "created_at": "2025-03-18T03:33:44.966Z",98 "last_posted_at": "2025-03-18T03:33:44.998Z",99 "bumped": true,100 "bumped_at": "2025-03-18T03:33:44.998Z",101 "archetype": "regular",102 "unseen": false,103 "pinned": false,104 "unpinned": null,105 "visible": true,106 "closed": false,107 "archived": false,108 "bookmarked": null,109 "liked": null,110 "tags_descriptions": {},111 "like_count": 0,112 "views": 63,113 "category_id": 8,114 "featured_link": null,115 "has_accepted_answer": false,116 "posters": [117 {118 "extras": "latest single",119 "description": "Original Poster, Most Recent Poster",120 "user": {121 "id": 83338,122 "username": "martin12",123 "name": "Martin",124 "avatar_template": "/user_avatar/discuss.pytorch.org/martin12/{size}/76222_2.png",125 "trust_level": 0126 }127 }128 ]129 },130 {131 "fancy_title": "A Simple LSTM stuck into label flipping",132 "id": 220004,133 "title": "A Simple LSTM stuck into label flipping",134 "slug": "a-simple-lstm-stuck-into-label-flipping",135 "posts_count": 5,136 "reply_count": 3,137 "highest_post_number": 5,138 "image_url": null,139 "created_at": "2025-05-13T19:44:43.099Z",140 "last_posted_at": "2025-05-15T08:28:26.193Z",141 "bumped": true,142 "bumped_at": "2025-05-15T08:28:26.193Z",143 "archetype": "regular",144 "unseen": false,145 "pinned": false,146 "unpinned": null,147 "visible": true,148 "closed": false,149 "archived": false,150 "bookmarked": null,151 "liked": null,152 "tags_descriptions": {},153 "like_count": 0,154 "views": 72,155 "category_id": 8,156 "featured_link": null,157 "has_accepted_answer": true,158 "posters": [159 {160 "extras": "latest",161 "description": "Original Poster, Most Recent Poster",162 "user": {163 "id": 84265,164 "username": "subhojitdas",165 "name": "Subhojit Das",166 "avatar_template": "/user_avatar/discuss.pytorch.org/subhojitdas/{size}/77001_2.png",167 "trust_level": 1168 }169 },170 {171 "extras": null,172 "description": "Frequent Poster, Accepted Answer",173 "user": {174 "id": 1438,175 "username": "vdw",176 "name": "Chris",177 "avatar_template": "/user_avatar/discuss.pytorch.org/vdw/{size}/10074_2.png",178 "trust_level": 2179 }180 }181 ]182 },183 {184 "fancy_title": "LSTM for classification (fraud detection) over several lines of text",185 "id": 216350,186 "title": "LSTM for classification (fraud detection) over several lines of text",187 "slug": "lstm-for-classification-fraud-detection-over-several-lines-of-text",188 "posts_count": 1,189 "reply_count": 0,190 "highest_post_number": 1,191 "image_url": null,192 "created_at": "2025-02-07T10:42:24.333Z",193 "last_posted_at": "2025-02-07T10:42:24.380Z",194 "bumped": true,195 "bumped_at": "2025-02-07T11:24:44.201Z",196 "archetype": "regular",197 "unseen": false,198 "pinned": false,199 "unpinned": null,200 "visible": true,201 "closed": false,202 "archived": false,203 "bookmarked": null,204 "liked": null,205 "tags_descriptions": {},206 "like_count": 0,207 "views": 144,208 "category_id": 8,209 "featured_link": null,210 "has_accepted_answer": false,211 "posters": [212 {213 "extras": "latest single",214 "description": "Original Poster, Most Recent Poster",215 "user": {216 "id": 82539,217 "username": "Monkee_Motion",218 "name": "Monkee Motion",219 "avatar_template": "/user_avatar/discuss.pytorch.org/monkee_motion/{size}/75523_2.png",220 "trust_level": 0221 }222 }223 ]224 },225 {226 "fancy_title": "How to compute the Validation loss",227 "id": 213365,228 "title": "How to compute the Validation loss",229 "slug": "how-to-compute-the-validation-loss",230 "posts_count": 3,231 "reply_count": 1,232 "highest_post_number": 3,233 "image_url": null,234 "created_at": "2024-11-24T08:21:30.769Z",235 "last_posted_at": "2024-11-24T16:57:48.437Z",236 "bumped": true,237 "bumped_at": "2024-11-24T16:57:48.437Z",238 "archetype": "regular",239 "unseen": false,240 "pinned": false,241 "unpinned": null,242 "visible": true,243 "closed": false,244 "archived": false,245 "bookmarked": null,246 "liked": null,247 "tags_descriptions": {},248 "like_count": 0,249 "views": 61,250 "category_id": 8,251 "featured_link": null,252 "has_accepted_answer": false,253 "posters": [254 {255 "extras": "latest",256 "description": "Original Poster, Most Recent Poster",257 "user": {258 "id": 81094,259 "username": "ali_fahad",260 "name": "ali fahad",261 "avatar_template": "/user_avatar/discuss.pytorch.org/ali_fahad/{size}/74160_2.png",262 "trust_level": 0263 }264 },265 {266 "extras": null,267 "description": "Frequent Poster",268 "user": {269 "id": 3534,270 "username": "ptrblck",271 "name": "",272 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",273 "admin": true,274 "moderator": true,275 "trust_level": 2276 }277 }278 ]279 },280 {281 "fancy_title": "Gemma 3 throws RuntimeError CUDA misaligned address",282 "id": 220507,283 "title": "Gemma 3 throws RuntimeError CUDA misaligned address",284 "slug": "gemma-3-throws-runtimeerror-cuda-misaligned-address",285 "posts_count": 2,286 "reply_count": 0,287 "highest_post_number": 2,288 "image_url": null,289 "created_at": "2025-06-02T07:00:32.299Z",290 "last_posted_at": "2025-06-03T22:25:43.492Z",291 "bumped": true,292 "bumped_at": "2025-06-03T22:25:43.492Z",293 "archetype": "regular",294 "unseen": false,295 "pinned": false,296 "unpinned": null,297 "visible": true,298 "closed": false,299 "archived": false,300 "bookmarked": null,301 "liked": null,302 "tags_descriptions": {},303 "like_count": 0,304 "views": 116,305 "category_id": 8,306 "featured_link": null,307 "has_accepted_answer": false,308 "posters": [309 {310 "extras": null,311 "description": "Original Poster",312 "user": {313 "id": 84544,314 "username": "msi-sbraun-11",315 "name": "",316 "avatar_template": "/user_avatar/discuss.pytorch.org/msi-sbraun-11/{size}/77240_2.png",317 "trust_level": 0318 }319 },320 {321 "extras": "latest",322 "description": "Most Recent Poster",323 "user": {324 "id": 3534,325 "username": "ptrblck",326 "name": "",327 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",328 "admin": true,329 "moderator": true,330 "trust_level": 2331 }332 }333 ]334 }335 ],336 "tags_descriptions": {},337 "fancy_title": "I trained same network with pytorch but totally bad results",338 "id": 81725,339 "title": "I trained same network with pytorch but totally bad results",340 "posts_count": 1,341 "created_at": "2020-05-18T02:30:14.640Z",342 "views": 357,343 "reply_count": 0,344 "like_count": 0,345 "last_posted_at": "2020-05-18T02:30:14.709Z",346 "visible": true,347 "closed": false,348 "archived": false,349 "has_summary": false,350 "archetype": "regular",351 "slug": "i-trained-same-network-with-pytorch-but-totally-bad-results",352 "category_id": 8,353 "word_count": 155,354 "deleted_at": null,355 "user_id": 30085,356 "featured_link": null,357 "pinned_globally": false,358 "pinned_at": null,359 "pinned_until": null,360 "image_url": null,361 "slow_mode_seconds": 0,362 "draft": null,363 "draft_key": "topic_81725",364 "draft_sequence": null,365 "unpinned": null,366 "pinned": false,367 "current_post_number": 1,368 "highest_post_number": 1,369 "deleted_by": null,370 "actions_summary": [371 {372 "id": 4,373 "count": 0,374 "hidden": false,375 "can_act": false376 },377 {378 "id": 8,379 "count": 0,380 "hidden": false,381 "can_act": false382 },383 {384 "id": 10,385 "count": 0,386 "hidden": false,387 "can_act": false388 },389 {390 "id": 7,391 "count": 0,392 "hidden": false,393 "can_act": false394 }395 ],396 "chunk_size": 20,397 "bookmarked": false,398 "topic_timer": null,399 "message_bus_last_id": 0,400 "participant_count": 1,401 "show_read_indicator": false,402 "thumbnails": null,403 "slow_mode_enabled_until": null,404 "can_vote": false,405 "vote_count": 0,406 "user_voted": false,407 "discourse_zendesk_plugin_zendesk_id": null,408 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",409 "details": {410 "can_edit": false,411 "notification_level": 1,412 "participants": [413 {414 "id": 30085,415 "username": "Ali_Amiri",416 "name": "Ali Amiri",417 "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png",418 "post_count": 1,419 "primary_group_name": null,420 "flair_name": null,421 "flair_url": null,422 "flair_color": null,423 "flair_bg_color": null,424 "flair_group_id": null,425 "trust_level": 1426 }427 ],428 "created_by": {429 "id": 30085,430 "username": "Ali_Amiri",431 "name": "Ali Amiri",432 "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png"433 },434 "last_poster": {435 "id": 30085,436 "username": "Ali_Amiri",437 "name": "Ali Amiri",438 "avatar_template": "/user_avatar/discuss.pytorch.org/ali_amiri/{size}/22823_2.png"439 },440 "links": [441 {442 "url": "https://colab.research.google.com/drive/1cz5Q-FgThpmM0lAT1mI8AgCRNi9ctsl-?usp=sharing",443 "title": null,444 "internal": false,445 "attachment": false,446 "reflection": false,447 "clicks": 1,448 "user_id": 30085,449 "domain": "colab.research.google.com",450 "root_domain": "google.com"451 },452 {453 "url": "https://colab.research.google.com/drive/1GlWjfKo9_GYz93rW4ZWyR6VSN45Bec3v?usp=sharing",454 "title": null,455 "internal": false,456 "attachment": false,457 "reflection": false,458 "clicks": 1,459 "user_id": 30085,460 "domain": "colab.research.google.com",461 "root_domain": "google.com"462 }463 ]464 },465 "bookmarks": []466 },467 {468 "post_stream": {469 "posts": [470 {471 "id": 193950,472 "name": "Shantanu Ghosh",473 "username": "Shantanu_Ghosh",474 "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png",475 "created_at": "2020-05-18T01:15:42.733Z",476 "cooked": "<p>Hi,<br>\nI have trained an autoencoder to get the lower dimensional representation of a image dataset.<br>\nNow I have saved the model. Then I load it and I want to remove the weights of the decoders. With the weights of the encoder I want to intiatlise the conv layers of a CNN classifier. This CNN classifier has same structure with the encoder of the autoencoder. Can anyone help me?</p>\n<p>Basically I want to implement this project(implemented in keras)in pytorch:<br>\n</p><aside class=\"onebox whitelistedgeneric\">\n <header class=\"source\">\n <img src=\"https://cdn.datacamp.com/main-app/assets/favicon-335cd0394b32102a39221d79e5fd7e51078e6d32a0c8aea59676a6869f84e9d8.ico\" class=\"site-icon\" width=\"64\" height=\"64\">\n <a href=\"https://www.datacamp.com/community/tutorials/autoencoder-classifier-python\" target=\"_blank\" rel=\"nofollow noopener\" title=\"04:00PM - 20 July 2018\">DataCamp Community – 20 Jul 18</a>\n </header>\n <article class=\"onebox-body\">\n <img src=\"https://cdn.datacamp.com/community/logos/medium/placeholder.jpg\" class=\"thumbnail onebox-avatar\" width=\"240\" height=\"240\">\n\n<h3><a href=\"https://www.datacamp.com/community/tutorials/autoencoder-classifier-python\" target=\"_blank\" rel=\"nofollow noopener\">Autoencoder as a Classifier</a></h3>\n\n<p>In this tutorial, you will learn & understand how to use autoencoder as a classifier in Python with Keras. You'll be using Fashion-MNIST dataset as an example.</p>\n\n\n </article>\n <div class=\"onebox-metadata\">\n \n \n </div>\n <div style=\"clear: both\"></div>\n</aside>\n",477 "post_number": 1,478 "post_type": 1,479 "posts_count": 2,480 "updated_at": "2020-05-18T02:57:06.123Z",481 "reply_count": 1,482 "reply_to_post_number": null,483 "quote_count": 0,484 "incoming_link_count": 150,485 "reads": 12,486 "readers_count": 11,487 "score": 757.4,488 "yours": false,489 "topic_id": 81713,490 "topic_slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",491 "display_username": "Shantanu Ghosh",492 "primary_group_name": null,493 "flair_name": null,494 "flair_url": null,495 "flair_bg_color": null,496 "flair_color": null,497 "flair_group_id": null,498 "badges_granted": [],499 "version": 3,500 "can_edit": false,501 "can_delete": false,502 "can_recover": false,503 "can_see_hidden_post": false,504 "can_wiki": false,505 "link_counts": [506 {507 "url": "https://www.datacamp.com/community/tutorials/autoencoder-classifier-python",508 "internal": false,509 "reflection": false,510 "title": "Autoencoder as a Classifier - DataCamp",511 "clicks": 10512 }513 ],514 "read": true,515 "user_title": null,516 "bookmarked": false,517 "actions_summary": [],518 "moderator": false,519 "admin": false,520 "staff": false,521 "user_id": 31776,522 "hidden": false,523 "trust_level": 1,524 "deleted_at": null,525 "user_deleted": false,526 "edit_reason": null,527 "can_view_edit_history": true,528 "wiki": false,529 "post_url": "/t/initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder/81713/1",530 "can_accept_answer": false,531 "can_unaccept_answer": false,532 "accepted_answer": false,533 "topic_accepted_answer": null,534 "can_vote": false535 },536 {537 "id": 193967,538 "name": "Jaya Krishna Mandivarapu",539 "username": "jmandivarapu1",540 "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png",541 "created_at": "2020-05-18T02:28:21.761Z",542 "cooked": "<aside class=\"quote no-group quote-modified\" data-username=\"Shantanu_Ghosh\" data-post=\"1\" data-topic=\"81713\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/user_avatar/discuss.pytorch.org/shantanu_ghosh/48/24369_2.png\" class=\"avatar\"> Shantanu_Ghosh:</div>\n<blockquote>\n<p>class CNN(nn.Module):<br>\ndef <strong>init</strong> (self):<br>\nsuper(CNN, self). <strong>init</strong> ()</p>\n<pre><code class=\"lang-auto\"> # Conv layers\n self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n self.bn1 = nn.BatchNorm2d(96)\n self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n self.bn2 = nn.BatchNorm2d(256)\n self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n self.pool = nn.MaxPool2d(3, 2)\n # FC layers\n</code></pre>\n</blockquote>\n</aside>\n<pre><code class=\"lang-auto\">class Autoencoder(nn.Module):\n def __init__(self):\n super(Autoencoder, self).__init__()\n # encoder layers\n self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n self.bn1 = nn.BatchNorm2d(96)\n self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n self.bn2 = nn.BatchNorm2d(256)\n self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n self.pool = nn.MaxPool2d(3, 2)\n\n # decoder layers\n self.dec1 = nn.ConvTranspose2d(384, 256, kernel_size=3, stride=2)\n self.dec2 = nn.ConvTranspose2d(256, 96, kernel_size=3, stride=2)\n self.dec3 = nn.ConvTranspose2d(96, 3, kernel_size=11, stride=4)\n\n def forward(self, x):\n\n # encoder\n x = F.relu(self.bn1(self.enc1(x)))\n x = self.pool(x)\n x = F.relu(self.bn2(self.enc2(x)))\n x = self.pool(x)\n x = F.relu(self.enc3(x)) # the latent space representation\n\n # decoder\n x = F.relu(self.dec1(x))\n x = F.relu(self.dec2(x))\n x = F.relu(self.dec3(x))\n # x = F.relu(self.dec4(x))\n # x = F.sigmoid(self.out(x))\n\n return x\nclass CNN(nn.Module):\n def __init__(self):\n super(CNN, self).__init__()\n\n # Conv layers\n self.enc1 = nn.Conv2d(3, 96, kernel_size=11, stride=4)\n self.bn1 = nn.BatchNorm2d(96)\n self.enc2 = nn.Conv2d(96, 256, kernel_size=5, stride=1, padding=2)\n self.bn2 = nn.BatchNorm2d(256)\n self.enc3 = nn.Conv2d(256, 384, kernel_size=3, stride=1, padding=1)\n self.pool = nn.MaxPool2d(3, 2)\n # FC layers\n \nx=Autoencoder()\ny=CNN()\nprint(x.enc1.weight)\nx.enc1.weight=y.enc1.weight\n\nprint(x.enc1.weight)\n</code></pre>",543 "post_number": 2,544 "post_type": 1,545 "posts_count": 2,546 "updated_at": "2020-05-18T02:28:21.761Z",547 "reply_count": 0,548 "reply_to_post_number": null,549 "quote_count": 1,550 "incoming_link_count": 10,551 "reads": 12,552 "readers_count": 11,553 "score": 52.4,554 "yours": false,555 "topic_id": 81713,556 "topic_slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",557 "display_username": "Jaya Krishna Mandivarapu",558 "primary_group_name": null,559 "flair_name": null,560 "flair_url": null,561 "flair_bg_color": null,562 "flair_color": null,563 "flair_group_id": null,564 "badges_granted": [],565 "version": 1,566 "can_edit": false,567 "can_delete": false,568 "can_recover": false,569 "can_see_hidden_post": false,570 "can_wiki": false,571 "read": true,572 "user_title": "",573 "bookmarked": false,574 "actions_summary": [],575 "moderator": false,576 "admin": false,577 "staff": false,578 "user_id": 3781,579 "hidden": false,580 "trust_level": 2,581 "deleted_at": null,582 "user_deleted": false,583 "edit_reason": null,584 "can_view_edit_history": true,585 "wiki": false,586 "post_url": "/t/initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder/81713/2",587 "can_accept_answer": false,588 "can_unaccept_answer": false,589 "accepted_answer": false,590 "topic_accepted_answer": null591 }592 ],593 "stream": [594 193950,595 193967596 ]597 },598 "timeline_lookup": [599 [600 1,601 1987602 ]603 ],604 "suggested_topics": [605 {606 "fancy_title": "Batch size at inference is influencing accuracy",607 "id": 218096,608 "title": "Batch size at inference is influencing accuracy",609 "slug": "batch-size-at-inference-is-influencing-accuracy",610 "posts_count": 1,611 "reply_count": 0,612 "highest_post_number": 1,613 "image_url": null,614 "created_at": "2025-03-20T21:02:58.000Z",615 "last_posted_at": "2025-03-20T21:02:58.052Z",616 "bumped": true,617 "bumped_at": "2025-03-20T21:02:58.052Z",618 "archetype": "regular",619 "unseen": false,620 "pinned": false,621 "unpinned": null,622 "visible": true,623 "closed": false,624 "archived": false,625 "bookmarked": null,626 "liked": null,627 "tags_descriptions": {},628 "like_count": 0,629 "views": 45,630 "category_id": 5,631 "featured_link": null,632 "has_accepted_answer": false,633 "posters": [634 {635 "extras": "latest single",636 "description": "Original Poster, Most Recent Poster",637 "user": {638 "id": 83393,639 "username": "danbull-scanabull",640 "name": "Danbull Scanabull",641 "avatar_template": "/user_avatar/discuss.pytorch.org/danbull-scanabull/{size}/76268_2.png",642 "trust_level": 1643 }644 }645 ]646 },647 {648 "fancy_title": "TypeError: ‘int’ object is not callable for claculating training accuracy",649 "id": 212609,650 "title": "TypeError: 'int' object is not callable for claculating training accuracy",651 "slug": "typeerror-int-object-is-not-callable-for-claculating-training-accuracy",652 "posts_count": 2,653 "reply_count": 0,654 "highest_post_number": 2,655 "image_url": null,656 "created_at": "2024-11-06T11:04:54.598Z",657 "last_posted_at": "2024-11-06T17:20:42.372Z",658 "bumped": true,659 "bumped_at": "2024-11-06T17:20:42.372Z",660 "archetype": "regular",661 "unseen": false,662 "pinned": false,663 "unpinned": null,664 "visible": true,665 "closed": false,666 "archived": false,667 "bookmarked": null,668 "liked": null,669 "tags_descriptions": {},670 "like_count": 0,671 "views": 30,672 "category_id": 5,673 "featured_link": null,674 "has_accepted_answer": false,675 "posters": [676 {677 "extras": null,678 "description": "Original Poster",679 "user": {680 "id": 68222,681 "username": "amy2",682 "name": "amy",683 "avatar_template": "/user_avatar/discuss.pytorch.org/amy2/{size}/62675_2.png",684 "trust_level": 1685 }686 },687 {688 "extras": "latest",689 "description": "Most Recent Poster",690 "user": {691 "id": 3534,692 "username": "ptrblck",693 "name": "",694 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",695 "admin": true,696 "moderator": true,697 "trust_level": 2698 }699 }700 ]701 },702 {703 "fancy_title": "Extracting Swin-Vit backbone",704 "id": 212799,705 "title": "Extracting Swin-Vit backbone",706 "slug": "extracting-swin-vit-backbone",707 "posts_count": 1,708 "reply_count": 0,709 "highest_post_number": 1,710 "image_url": null,711 "created_at": "2024-11-11T09:54:48.491Z",712 "last_posted_at": "2024-11-11T09:54:48.540Z",713 "bumped": true,714 "bumped_at": "2024-11-11T09:54:48.540Z",715 "archetype": "regular",716 "unseen": false,717 "pinned": false,718 "unpinned": null,719 "visible": true,720 "closed": false,721 "archived": false,722 "bookmarked": null,723 "liked": null,724 "tags_descriptions": {},725 "like_count": 0,726 "views": 196,727 "category_id": 5,728 "featured_link": null,729 "has_accepted_answer": false,730 "posters": [731 {732 "extras": "latest single",733 "description": "Original Poster, Most Recent Poster",734 "user": {735 "id": 56688,736 "username": "ima",737 "name": "Imantha gunasekera",738 "avatar_template": "/letter_avatar_proxy/v4/letter/i/67e7ee/{size}.png",739 "trust_level": 1740 }741 }742 ]743 },744 {745 "fancy_title": "Semantic Segmentation with Attention based CycleGAN",746 "id": 217647,747 "title": "Semantic Segmentation with Attention based CycleGAN",748 "slug": "semantic-segmentation-with-attention-based-cyclegan",749 "posts_count": 5,750 "reply_count": 1,751 "highest_post_number": 5,752 "image_url": null,753 "created_at": "2025-03-10T06:44:59.887Z",754 "last_posted_at": "2025-03-11T22:57:33.781Z",755 "bumped": true,756 "bumped_at": "2025-03-11T22:57:33.781Z",757 "archetype": "regular",758 "unseen": false,759 "pinned": false,760 "unpinned": null,761 "visible": true,762 "closed": false,763 "archived": false,764 "bookmarked": null,765 "liked": null,766 "tags_descriptions": {},767 "like_count": 0,768 "views": 133,769 "category_id": 5,770 "featured_link": null,771 "has_accepted_answer": false,772 "posters": [773 {774 "extras": null,775 "description": "Original Poster",776 "user": {777 "id": 81422,778 "username": "Idrees11",779 "name": "Idrees Bhat",780 "avatar_template": "/user_avatar/discuss.pytorch.org/idrees11/{size}/74448_2.png",781 "trust_level": 1782 }783 },784 {785 "extras": "latest",786 "description": "Most Recent Poster",787 "user": {788 "id": 18088,789 "username": "KFrank",790 "name": "K. Frank",791 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",792 "trust_level": 2793 }794 }795 ]796 },797 {798 "fancy_title": "My Loss function becomes 0 in 2nd epoch",799 "id": 221317,800 "title": "My Loss function becomes 0 in 2nd epoch",801 "slug": "my-loss-function-becomes-0-in-2nd-epoch",802 "posts_count": 2,803 "reply_count": 0,804 "highest_post_number": 2,805 "image_url": null,806 "created_at": "2025-07-07T05:55:40.826Z",807 "last_posted_at": "2025-07-08T22:59:30.226Z",808 "bumped": true,809 "bumped_at": "2025-07-08T22:59:30.226Z",810 "archetype": "regular",811 "unseen": false,812 "pinned": false,813 "unpinned": null,814 "visible": true,815 "closed": false,816 "archived": false,817 "bookmarked": null,818 "liked": null,819 "tags_descriptions": {},820 "like_count": 1,821 "views": 74,822 "category_id": 5,823 "featured_link": null,824 "has_accepted_answer": false,825 "posters": [826 {827 "extras": null,828 "description": "Original Poster",829 "user": {830 "id": 84964,831 "username": "syeda_raheen",832 "name": "Raheen",833 "avatar_template": "/letter_avatar_proxy/v4/letter/s/b19c9b/{size}.png",834 "trust_level": 0835 }836 },837 {838 "extras": "latest",839 "description": "Most Recent Poster",840 "user": {841 "id": 84484,842 "username": "Dhia-naouali",843 "name": "Dhia naouali",844 "avatar_template": "/user_avatar/discuss.pytorch.org/dhia-naouali/{size}/77193_2.png",845 "trust_level": 2846 }847 }848 ]849 }850 ],851 "tags_descriptions": {},852 "fancy_title": "Initialise the weights of the conv layers with the weights of a pretrained autoencoder",853 "id": 81713,854 "title": "Initialise the weights of the conv layers with the weights of a pretrained autoencoder",855 "posts_count": 2,856 "created_at": "2020-05-18T01:15:42.644Z",857 "views": 545,858 "reply_count": 0,859 "like_count": 0,860 "last_posted_at": "2020-05-18T02:28:21.761Z",861 "visible": true,862 "closed": false,863 "archived": false,864 "has_summary": false,865 "archetype": "regular",866 "slug": "initialise-the-weights-of-the-conv-layers-with-the-weights-of-a-pretrained-autoencoder",867 "category_id": 5,868 "word_count": 412,869 "deleted_at": null,870 "user_id": 31776,871 "featured_link": null,872 "pinned_globally": false,873 "pinned_at": null,874 "pinned_until": null,875 "image_url": null,876 "slow_mode_seconds": 0,877 "draft": null,878 "draft_key": "topic_81713",879 "draft_sequence": null,880 "unpinned": null,881 "pinned": false,882 "current_post_number": 1,883 "highest_post_number": 2,884 "deleted_by": null,885 "actions_summary": [886 {887 "id": 4,888 "count": 0,889 "hidden": false,890 "can_act": false891 },892 {893 "id": 8,894 "count": 0,895 "hidden": false,896 "can_act": false897 },898 {899 "id": 10,900 "count": 0,901 "hidden": false,902 "can_act": false903 },904 {905 "id": 7,906 "count": 0,907 "hidden": false,908 "can_act": false909 }910 ],911 "chunk_size": 20,912 "bookmarked": false,913 "topic_timer": null,914 "message_bus_last_id": 0,915 "participant_count": 2,916 "show_read_indicator": false,917 "thumbnails": null,918 "slow_mode_enabled_until": null,919 "can_vote": false,920 "vote_count": 0,921 "user_voted": false,922 "discourse_zendesk_plugin_zendesk_id": null,923 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",924 "details": {925 "can_edit": false,926 "notification_level": 1,927 "participants": [928 {929 "id": 3781,930 "username": "jmandivarapu1",931 "name": "Jaya Krishna Mandivarapu",932 "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png",933 "post_count": 1,934 "primary_group_name": null,935 "flair_name": null,936 "flair_url": null,937 "flair_color": null,938 "flair_bg_color": null,939 "flair_group_id": null,940 "trust_level": 2941 },942 {943 "id": 31776,944 "username": "Shantanu_Ghosh",945 "name": "Shantanu Ghosh",946 "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png",947 "post_count": 1,948 "primary_group_name": null,949 "flair_name": null,950 "flair_url": null,951 "flair_color": null,952 "flair_bg_color": null,953 "flair_group_id": null,954 "trust_level": 1955 }956 ],957 "created_by": {958 "id": 31776,959 "username": "Shantanu_Ghosh",960 "name": "Shantanu Ghosh",961 "avatar_template": "/user_avatar/discuss.pytorch.org/shantanu_ghosh/{size}/24369_2.png"962 },963 "last_poster": {964 "id": 3781,965 "username": "jmandivarapu1",966 "name": "Jaya Krishna Mandivarapu",967 "avatar_template": "/user_avatar/discuss.pytorch.org/jmandivarapu1/{size}/3544_2.png"968 },969 "links": [970 {971 "url": "https://www.datacamp.com/community/tutorials/autoencoder-classifier-python",972 "title": "Autoencoder as a Classifier - DataCamp",973 "internal": false,974 "attachment": false,975 "reflection": false,976 "clicks": 10,977 "user_id": 31776,978 "domain": "www.datacamp.com",979 "root_domain": "datacamp.com"980 }981 ]982 },983 "bookmarks": []984 },985 {986 "post_stream": {987 "posts": [988 {989 "id": 192846,990 "name": "Zach",991 "username": "yangkz",992 "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",993 "created_at": "2020-05-14T18:10:46.029Z",994 "cooked": "<p>When I use two gpus to train my model, I got RuntimeError below:</p>\n<p>Process SpawnProcess-2:<br>\nTraceback (most recent call last):<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 258, in _bootstrap<br>\nself.run()<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 93, in run<br>\nself._target(*self._args, **self._kwargs)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 190, in run<br>\nmain(rank, dev_id, args)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 149, in main<br>\ntrain(args[‘gnn’], model, device, train_loader, criterion, optimizer, args[‘num_devices’], rank)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 41, in train<br>\noptimizer.backward_and_step(loss)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 146, in backward_and_step<br>\nself._sync_gradient()<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 127, in _sync_gradient<br>\ndist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/distributed_c10d.py”, line 902, in all_reduce<br>\nwork = _default_pg.allreduce([tensor], opts)<br>\nRuntimeError: Stop_waiting response is expected</p>\n<p>Process SpawnProcess-1:<br>\nTraceback (most recent call last):<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 258, in _bootstrap<br>\nself.run()<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/multiprocessing/process.py”, line 93, in run<br>\nself._target(*self._args, **self._kwargs)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 190, in run<br>\nmain(rank, dev_id, args)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 149, in main<br>\ntrain(args[‘gnn’], model, device, train_loader, criterion, optimizer, args[‘num_devices’], rank)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/m.py”, line 41, in train<br>\noptimizer.backward_and_step(loss)<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 146, in backward_and_step<br>\nself._sync_gradient()<br>\nFile “/home/ubuntu/ogb/ogb/graphproppred/utils.py”, line 127, in _sync_gradient<br>\ndist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)<br>\nFile “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/distributed_c10d.py”, line 902, in all_reduce<br>\nwork = _default_pg.allreduce([tensor], opts)<br>\nRuntimeError: Stop_waiting response is expected</p>\n<p>Here is the code where the error occurred:</p>\n<pre><code>def _sync_gradient(self):\n \"\"\"Average gradients across all subprocesses.\"\"\"\n for param_group in self.optimizer.param_groups:\n for p in param_group['params']:\n if p.requires_grad and p.grad is not None:\n # print(p.grad.data.shape, p.grad.data.device)\n dist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)\n p.grad.data /= self.n_processes\n</code></pre>\n<p>Ps. When I do “print(p.grad.data.shape, p.grad.data.device)”, I find the grads are normal and has the same shape [1,300] on 2 different gpus. So I’m confused why it stopped here.</p>",995 "post_number": 1,996 "post_type": 1,997 "posts_count": 4,998 "updated_at": "2020-05-14T18:10:46.029Z",999 "reply_count": 0,1000 "reply_to_post_number": null,1001 "quote_count": 0,1002 "incoming_link_count": 1902,1003 "reads": 28,1004 "readers_count": 27,1005 "score": 9515.6,1006 "yours": false,1007 "topic_id": 81248,1008 "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1009 "display_username": "Zach",1010 "primary_group_name": null,1011 "flair_name": null,1012 "flair_url": null,1013 "flair_bg_color": null,1014 "flair_color": null,1015 "flair_group_id": null,1016 "badges_granted": [],1017 "version": 1,1018 "can_edit": false,1019 "can_delete": false,1020 "can_recover": false,1021 "can_see_hidden_post": false,1022 "can_wiki": false,1023 "read": true,1024 "user_title": "",1025 "bookmarked": false,1026 "actions_summary": [],1027 "moderator": false,1028 "admin": false,1029 "staff": false,1030 "user_id": 31619,1031 "hidden": false,1032 "trust_level": 1,1033 "deleted_at": null,1034 "user_deleted": false,1035 "edit_reason": null,1036 "can_view_edit_history": true,1037 "wiki": false,1038 "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/1",1039 "can_accept_answer": false,1040 "can_unaccept_answer": false,1041 "accepted_answer": false,1042 "topic_accepted_answer": true,1043 "can_vote": false1044 },1045 {1046 "id": 192854,1047 "name": "Shen Li",1048 "username": "mrshenli",1049 "avatar_template": "/user_avatar/discuss.pytorch.org/mrshenli/{size}/12220_2.png",1050 "created_at": "2020-05-14T19:07:41.991Z",1051 "cooked": "<p>Is the result of <code>p.requires_grad and p.grad is not None</code> always the same across all process and all parameters? If allreduce ops on different processes could run into desync.</p>\n<p>Which backend are you using (NCCL/Gloo/MPI) and which PyTorch version are you using? It will be helpful to have a min repro of this error.</p>",1052 "post_number": 2,1053 "post_type": 1,1054 "posts_count": 4,1055 "updated_at": "2020-05-14T19:07:41.991Z",1056 "reply_count": 1,1057 "reply_to_post_number": null,1058 "quote_count": 0,1059 "incoming_link_count": 6,1060 "reads": 28,1061 "readers_count": 27,1062 "score": 55.6,1063 "yours": false,1064 "topic_id": 81248,1065 "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1066 "display_username": "Shen Li",1067 "primary_group_name": null,1068 "flair_name": null,1069 "flair_url": null,1070 "flair_bg_color": null,1071 "flair_color": null,1072 "flair_group_id": null,1073 "badges_granted": [],1074 "version": 1,1075 "can_edit": false,1076 "can_delete": false,1077 "can_recover": false,1078 "can_see_hidden_post": false,1079 "can_wiki": false,1080 "read": true,1081 "user_title": null,1082 "bookmarked": false,1083 "actions_summary": [1084 {1085 "id": 2,1086 "count": 11087 }1088 ],1089 "moderator": false,1090 "admin": false,1091 "staff": false,1092 "user_id": 17068,1093 "hidden": false,1094 "trust_level": 2,1095 "deleted_at": null,1096 "user_deleted": false,1097 "edit_reason": null,1098 "can_view_edit_history": true,1099 "wiki": false,1100 "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/2",1101 "can_accept_answer": false,1102 "can_unaccept_answer": false,1103 "accepted_answer": false,1104 "topic_accepted_answer": true1105 },1106 {1107 "id": 193074,1108 "name": "Zach",1109 "username": "yangkz",1110 "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",1111 "created_at": "2020-05-15T07:55:04.015Z",1112 "cooked": "<p>Thanks for your reply!<br>\nHere is my short code:</p>\n<pre><code class=\"lang-auto\">def main(rank, dev_id, args):\n torch.distributed.init_process_group(backend=\"nccl\",\n init_method='tcp://localhost:22',\n world_size=args['num_devices'],\n rank=dev_id)\n model = mymodel.to(dev_id)\n optimizer = optim.Adam(model.parameters(), lr=args['lr'])\n for epochs:\n pred = model(inputs)\n loss = criterion(pred, label)\n optimizer.zero_grad()\n loss.backward()\n\n for param_group in optimizer.param_groups:\n for p in param_group['params']:\n if p.requires_grad and p.grad is not None:\n # print(p.grad.data.shape, p.grad.data.device) Ps. We can get grad information here\n dist.all_reduce(p.grad.data, op=dist.ReduceOp.SUM)\n p.grad.data /= n_processes\n optimizer.step()\n torch.distributed.barrier()\n\nmp = torch.multiprocessing.get_context('spawn')\nfor id, device_id in enumerate(devices):\n procs.append(mp.Process(target=main, args=(id, device_id, args), daemon=True))\n procs[-1].start()\nfor p in procs:\n p.join()\n</code></pre>\n<p>The pytorch version I’m using is 1.4.0 and all my codes are running in AWS instance.<br>\nPs. The port of init_method can only be 22 which is strange otherwise it got Runtime Error like this:</p>\n<p>File “/home/ubuntu/anaconda3/envs/pytorch_p36/lib/python3.6/site-packages/torch/distributed/rendezvous.py”, line 120, in _tcp_rendezvous_handler<br>\nstore = TCPStore(result.hostname, result.port, world_size, start_daemon)<br>\nRuntimeError: connect() timed out.</p>\n<p>Thanks for your time and reading!</p>",1113 "post_number": 4,1114 "post_type": 1,1115 "posts_count": 4,1116 "updated_at": "2020-05-15T08:02:48.256Z",1117 "reply_count": 1,1118 "reply_to_post_number": 2,1119 "quote_count": 0,1120 "incoming_link_count": 160,1121 "reads": 27,1122 "readers_count": 26,1123 "score": 810.4,1124 "yours": false,1125 "topic_id": 81248,1126 "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1127 "display_username": "Zach",1128 "primary_group_name": null,1129 "flair_name": null,1130 "flair_url": null,1131 "flair_bg_color": null,1132 "flair_color": null,1133 "flair_group_id": null,1134 "badges_granted": [],1135 "version": 2,1136 "can_edit": false,1137 "can_delete": false,1138 "can_recover": false,1139 "can_see_hidden_post": false,1140 "can_wiki": false,1141 "read": true,1142 "user_title": "",1143 "reply_to_user": {1144 "id": 17068,1145 "username": "mrshenli",1146 "name": "Shen Li",1147 "avatar_template": "/user_avatar/discuss.pytorch.org/mrshenli/{size}/12220_2.png"1148 },1149 "bookmarked": false,1150 "actions_summary": [],1151 "moderator": false,1152 "admin": false,1153 "staff": false,1154 "user_id": 31619,1155 "hidden": false,1156 "trust_level": 1,1157 "deleted_at": null,1158 "user_deleted": false,1159 "edit_reason": null,1160 "can_view_edit_history": true,1161 "wiki": false,1162 "post_url": "/t/runtimeerror-stop-waiting-response-is-expected/81248/4",1163 "can_accept_answer": false,1164 "can_unaccept_answer": false,1165 "accepted_answer": false,1166 "topic_accepted_answer": true1167 },1168 {1169 "id": 193958,1170 "name": "Zach",1171 "username": "yangkz",1172 "avatar_template": "/letter_avatar_proxy/v4/letter/y/ea5d25/{size}.png",1173 "created_at": "2020-05-18T01:46:31.076Z",1174 "cooked": "<p>The error has been fixed.<br>\n‘Stop_waiting response is expected’ error occurred in TCPStore.cpp. So it was actually the communication problem. It works finally when I reinstalled NCCL: <a href=\"https://github.com/NVIDIA/nccl.git\" rel=\"nofollow noopener\">https://github.com/NVIDIA/nccl.git</a></p>",1175 "post_number": 5,1176 "post_type": 1,1177 "posts_count": 4,1178 "updated_at": "2020-05-18T01:46:35.425Z",1179 "reply_count": 0,1180 "reply_to_post_number": 4,1181 "quote_count": 0,1182 "incoming_link_count": 21,1183 "reads": 26,1184 "readers_count": 25,1185 "score": 110.2,1186 "yours": false,1187 "topic_id": 81248,1188 "topic_slug": "runtimeerror-stop-waiting-response-is-expected",1189 "display_username": "Zach",1190 "primary_group_name": null,1191 "flair_name": null,1192 "flair_url": null,1193 "flair_bg_color": null,1194 "flair_color": null,1195 "flair_group_id": null,1196 "badges_granted": [],1197 "version": 1,1198 "can_edit": false,1199 "can_delete": false,1200 "can_recover": false,