CoolFace
Datasetpublic

Anurag1734/cuda-error-resolution-analysis

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes7downloads
topics_batch_154.json61985 linesDownload Raw Back to raw
1[2  {3    "post_stream": {4      "posts": [5        {6          "id": 392788,7          "name": "Rob",8          "username": "RobS",9          "avatar_template": "/letter_avatar_proxy/v4/letter/r/c89c15/{size}.png",10          "created_at": "2023-03-18T02:55:46.443Z",11          "cooked": "<p>Docs here: <a href=\"https://pytorch.org/docs/stable/elastic/run.html#single-node-multi-worker\" class=\"inline-onebox\" rel=\"noopener nofollow ugc\">torchrun (Elastic Launch) — PyTorch 2.0 documentation</a></p>\n<p>In the Pytorch docs for torchrun, it lists two options for single-node multi-worker training: “Single-node multi-worker” and “Stacked single-node multi-worker”.</p>\n<p>For me the “single-node multi-worker” did not work as intended but the “Stacked single-node multi-worker” training worked exactly as expected. “single-node multi-worker” training only processed on one GPU while the “stacked” version of the command engaged all available GPUs. What are the intended differences and use cases of these two options?</p>",12          "post_number": 1,13          "post_type": 1,14          "posts_count": 1,15          "updated_at": "2023-03-18T02:55:46.443Z",16          "reply_count": 0,17          "reply_to_post_number": null,18          "quote_count": 0,19          "incoming_link_count": 36,20          "reads": 6,21          "readers_count": 5,22          "score": 181.2,23          "yours": false,24          "topic_id": 175198,25          "topic_slug": "stacked-vs-eponymous-torchrun-cli-options",26          "display_username": "Rob",27          "primary_group_name": null,28          "flair_name": null,29          "flair_url": null,30          "flair_bg_color": null,31          "flair_color": null,32          "flair_group_id": null,33          "badges_granted": [],34          "version": 1,35          "can_edit": false,36          "can_delete": false,37          "can_recover": false,38          "can_see_hidden_post": false,39          "can_wiki": false,40          "link_counts": [41            {42              "url": "https://pytorch.org/docs/stable/elastic/run.html#single-node-multi-worker",43              "internal": false,44              "reflection": false,45              "title": "torchrun (Elastic Launch) — PyTorch 2.0 documentation",46              "clicks": 247            }48          ],49          "read": true,50          "user_title": null,51          "bookmarked": false,52          "actions_summary": [],53          "moderator": false,54          "admin": false,55          "staff": false,56          "user_id": 64354,57          "hidden": false,58          "trust_level": 1,59          "deleted_at": null,60          "user_deleted": false,61          "edit_reason": null,62          "can_view_edit_history": true,63          "wiki": false,64          "post_url": "/t/stacked-vs-eponymous-torchrun-cli-options/175198/1",65          "can_accept_answer": false,66          "can_unaccept_answer": false,67          "accepted_answer": false,68          "topic_accepted_answer": null,69          "can_vote": false70        }71      ],72      "stream": [73        39278874      ]75    },76    "timeline_lookup": [77      [78        1,79        95380      ]81    ],82    "suggested_topics": [83      {84        "fancy_title": "Get_backend() returns undefined even when NCCL is available",85        "id": 219886,86        "title": "Get_backend() returns undefined even when NCCL is available",87        "slug": "get-backend-returns-undefined-even-when-nccl-is-available",88        "posts_count": 4,89        "reply_count": 1,90        "highest_post_number": 4,91        "image_url": null,92        "created_at": "2025-05-09T10:01:13.685Z",93        "last_posted_at": "2025-05-12T03:44:57.999Z",94        "bumped": true,95        "bumped_at": "2025-05-12T03:44:57.999Z",96        "archetype": "regular",97        "unseen": false,98        "pinned": false,99        "unpinned": null,100        "visible": true,101        "closed": false,102        "archived": false,103        "bookmarked": null,104        "liked": null,105        "tags_descriptions": {},106        "like_count": 4,107        "views": 142,108        "category_id": 12,109        "featured_link": null,110        "has_accepted_answer": true,111        "posters": [112          {113            "extras": "latest",114            "description": "Original Poster, Most Recent Poster",115            "user": {116              "id": 83712,117              "username": "joshkim",118              "name": null,119              "avatar_template": "/letter_avatar_proxy/v4/letter/j/a8b319/{size}.png",120              "trust_level": 1121            }122          },123          {124            "extras": null,125            "description": "Frequent Poster, Accepted Answer",126            "user": {127              "id": 55215,128              "username": "kwen2501",129              "name": "Ke Wen",130              "avatar_template": "/user_avatar/discuss.pytorch.org/kwen2501/{size}/48844_2.png",131              "trust_level": 1132            }133          },134          {135            "extras": null,136            "description": "Frequent Poster",137            "user": {138              "id": 3534,139              "username": "ptrblck",140              "name": "",141              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",142              "admin": true,143              "moderator": true,144              "trust_level": 2145            }146          }147        ]148      },149      {150        "fancy_title": "Why does init_device_mesh() or DeviceMesh() have to be called globally?",151        "id": 221457,152        "title": "Why does init_device_mesh() or DeviceMesh() have to be called globally?",153        "slug": "why-does-init-device-mesh-or-devicemesh-have-to-be-called-globally",154        "posts_count": 4,155        "reply_count": 0,156        "highest_post_number": 5,157        "image_url": null,158        "created_at": "2025-07-12T01:56:35.520Z",159        "last_posted_at": "2025-07-22T07:32:43.980Z",160        "bumped": true,161        "bumped_at": "2025-07-22T07:32:43.980Z",162        "archetype": "regular",163        "unseen": false,164        "pinned": false,165        "unpinned": null,166        "visible": true,167        "closed": false,168        "archived": false,169        "bookmarked": null,170        "liked": null,171        "tags_descriptions": {},172        "like_count": 0,173        "views": 79,174        "category_id": 12,175        "featured_link": null,176        "has_accepted_answer": false,177        "posters": [178          {179            "extras": null,180            "description": "Original Poster",181            "user": {182              "id": 82804,183              "username": "githubsgi",184              "name": "_githubsgi",185              "avatar_template": "/user_avatar/discuss.pytorch.org/githubsgi/{size}/75767_2.png",186              "trust_level": 1187            }188          },189          {190            "extras": null,191            "description": "Frequent Poster",192            "user": {193              "id": 6225,194              "username": "yf225",195              "name": "PyTorch Developer, Meta",196              "avatar_template": "/user_avatar/discuss.pytorch.org/yf225/{size}/3418_2.png",197              "trust_level": 2198            }199          },200          {201            "extras": null,202            "description": "Frequent Poster",203            "user": {204              "id": 54320,205              "username": "fduwjj",206              "name": "Hugo",207              "avatar_template": "/user_avatar/discuss.pytorch.org/fduwjj/{size}/47855_2.png",208              "trust_level": 2209            }210          },211          {212            "extras": "latest",213            "description": "Most Recent Poster",214            "user": {215              "id": 62488,216              "username": "irisz",217              "name": "Iris Z",218              "avatar_template": "/user_avatar/discuss.pytorch.org/irisz/{size}/56395_2.png",219              "trust_level": 2220            }221          }222        ]223      },224      {225        "fancy_title": "Torch.multiprocessing: pass metadata or class wrapper for shared memory CUDA tensor?",226        "id": 216101,227        "title": "Torch.multiprocessing: pass metadata or class wrapper for shared memory CUDA tensor?",228        "slug": "torch-multiprocessing-pass-metadata-or-class-wrapper-for-shared-memory-cuda-tensor",229        "posts_count": 1,230        "reply_count": 0,231        "highest_post_number": 1,232        "image_url": null,233        "created_at": "2025-01-31T18:30:16.330Z",234        "last_posted_at": "2025-01-31T18:30:16.375Z",235        "bumped": true,236        "bumped_at": "2025-01-31T18:30:16.375Z",237        "archetype": "regular",238        "unseen": false,239        "pinned": false,240        "unpinned": null,241        "visible": true,242        "closed": false,243        "archived": false,244        "bookmarked": null,245        "liked": null,246        "tags_descriptions": {},247        "like_count": 0,248        "views": 23,249        "category_id": 12,250        "featured_link": null,251        "has_accepted_answer": false,252        "posters": [253          {254            "extras": "latest single",255            "description": "Original Poster, Most Recent Poster",256            "user": {257              "id": 61091,258              "username": "pierisk",259              "name": "",260              "avatar_template": "/user_avatar/discuss.pytorch.org/pierisk/{size}/54972_2.png",261              "trust_level": 1262            }263          }264        ]265      },266      {267        "fancy_title": "Will doing two times forward and backward work fine?",268        "id": 218388,269        "title": "Will doing two times forward and backward work fine?",270        "slug": "will-doing-two-times-forward-and-backward-work-fine",271        "posts_count": 2,272        "reply_count": 0,273        "highest_post_number": 2,274        "image_url": null,275        "created_at": "2025-03-29T03:56:38.060Z",276        "last_posted_at": "2025-03-29T23:29:13.394Z",277        "bumped": true,278        "bumped_at": "2025-03-29T23:29:13.394Z",279        "archetype": "regular",280        "unseen": false,281        "pinned": false,282        "unpinned": null,283        "visible": true,284        "closed": false,285        "archived": false,286        "bookmarked": null,287        "liked": null,288        "tags_descriptions": {},289        "like_count": 1,290        "views": 40,291        "category_id": 12,292        "featured_link": null,293        "has_accepted_answer": true,294        "posters": [295          {296            "extras": null,297            "description": "Original Poster",298            "user": {299              "id": 23879,300              "username": "chener",301              "name": "chener",302              "avatar_template": "/user_avatar/discuss.pytorch.org/chener/{size}/17159_2.png",303              "trust_level": 1304            }305          },306          {307            "extras": "latest",308            "description": "Most Recent Poster, Accepted Answer",309            "user": {310              "id": 18088,311              "username": "KFrank",312              "name": "K. Frank",313              "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",314              "trust_level": 2315            }316          }317        ]318      },319      {320        "fancy_title": "Variable batch size in Multi-GPU trainings",321        "id": 221940,322        "title": "Variable batch size in Multi-GPU trainings",323        "slug": "variable-batch-size-in-multi-gpu-trainings",324        "posts_count": 4,325        "reply_count": 2,326        "highest_post_number": 4,327        "image_url": null,328        "created_at": "2025-07-30T05:10:20.832Z",329        "last_posted_at": "2025-07-31T14:59:57.708Z",330        "bumped": true,331        "bumped_at": "2025-07-31T14:59:57.708Z",332        "archetype": "regular",333        "unseen": false,334        "pinned": false,335        "unpinned": null,336        "visible": true,337        "closed": false,338        "archived": false,339        "bookmarked": null,340        "liked": null,341        "tags_descriptions": {},342        "like_count": 0,343        "views": 54,344        "category_id": 12,345        "featured_link": null,346        "has_accepted_answer": false,347        "posters": [348          {349            "extras": null,350            "description": "Original Poster",351            "user": {352              "id": 85290,353              "username": "Ryan_Xu",354              "name": "Ryan Xu",355              "avatar_template": "/user_avatar/discuss.pytorch.org/ryan_xu/{size}/77825_2.png",356              "trust_level": 1357            }358          },359          {360            "extras": "latest",361            "description": "Most Recent Poster",362            "user": {363              "id": 3534,364              "username": "ptrblck",365              "name": "",366              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",367              "admin": true,368              "moderator": true,369              "trust_level": 2370            }371          }372        ]373      }374    ],375    "tags_descriptions": {},376    "fancy_title": "Stacked vs. eponymous torchrun cli options",377    "id": 175198,378    "title": "Stacked vs. eponymous torchrun cli options",379    "posts_count": 1,380    "created_at": "2023-03-18T02:55:46.363Z",381    "views": 409,382    "reply_count": 0,383    "like_count": 0,384    "last_posted_at": "2023-03-18T02:55:46.443Z",385    "visible": true,386    "closed": false,387    "archived": false,388    "has_summary": false,389    "archetype": "regular",390    "slug": "stacked-vs-eponymous-torchrun-cli-options",391    "category_id": 12,392    "word_count": 97,393    "deleted_at": null,394    "user_id": 64354,395    "featured_link": null,396    "pinned_globally": false,397    "pinned_at": null,398    "pinned_until": null,399    "image_url": null,400    "slow_mode_seconds": 0,401    "draft": null,402    "draft_key": "topic_175198",403    "draft_sequence": null,404    "unpinned": null,405    "pinned": false,406    "current_post_number": 1,407    "highest_post_number": 1,408    "deleted_by": null,409    "actions_summary": [410      {411        "id": 4,412        "count": 0,413        "hidden": false,414        "can_act": false415      },416      {417        "id": 8,418        "count": 0,419        "hidden": false,420        "can_act": false421      },422      {423        "id": 10,424        "count": 0,425        "hidden": false,426        "can_act": false427      },428      {429        "id": 7,430        "count": 0,431        "hidden": false,432        "can_act": false433      }434    ],435    "chunk_size": 20,436    "bookmarked": false,437    "topic_timer": null,438    "message_bus_last_id": 0,439    "participant_count": 1,440    "show_read_indicator": false,441    "thumbnails": null,442    "slow_mode_enabled_until": null,443    "can_vote": false,444    "vote_count": 0,445    "user_voted": false,446    "discourse_zendesk_plugin_zendesk_id": null,447    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",448    "details": {449      "can_edit": false,450      "notification_level": 1,451      "participants": [452        {453          "id": 64354,454          "username": "RobS",455          "name": "Rob",456          "avatar_template": "/letter_avatar_proxy/v4/letter/r/c89c15/{size}.png",457          "post_count": 1,458          "primary_group_name": null,459          "flair_name": null,460          "flair_url": null,461          "flair_color": null,462          "flair_bg_color": null,463          "flair_group_id": null,464          "trust_level": 1465        }466      ],467      "created_by": {468        "id": 64354,469        "username": "RobS",470        "name": "Rob",471        "avatar_template": "/letter_avatar_proxy/v4/letter/r/c89c15/{size}.png"472      },473      "last_poster": {474        "id": 64354,475        "username": "RobS",476        "name": "Rob",477        "avatar_template": "/letter_avatar_proxy/v4/letter/r/c89c15/{size}.png"478      },479      "links": [480        {481          "url": "https://pytorch.org/docs/stable/elastic/run.html#single-node-multi-worker",482          "title": "torchrun (Elastic Launch) — PyTorch 2.0 documentation",483          "internal": false,484          "attachment": false,485          "reflection": false,486          "clicks": 2,487          "user_id": 64354,488          "domain": "pytorch.org",489          "root_domain": "pytorch.org"490        }491      ]492    },493    "bookmarks": []494  },495  {496    "post_stream": {497      "posts": [498        {499          "id": 392730,500          "name": "Samster",501          "username": "papoo13",502          "avatar_template": "/letter_avatar_proxy/v4/letter/p/f19dbf/{size}.png",503          "created_at": "2023-03-17T15:21:08.118Z",504          "cooked": "<p>I face some instability in training my model in a continual setting. I have 20 tasks and after task 12, the accuracy goes down to 0.0007, so nothing happens in terms of learning. My hypothesis is maybe the problem is a numerical instability. so, I would like to train on a higher precision, aka float64. To so so, I am changing my code as follows:</p>\n<pre><code class=\"lang-auto\">def update_model(self, x, y, criterion, optimizer):\n        # chekc the label type, output of the bayesian model\n        \n        optimizer.zero_grad()\n        do_cutmix = self.cutmix and np.random.rand(1) &lt; 0.5\n        if do_cutmix:\n            x, labels_a, labels_b, lam = cutmix_data(x=x, y=y, alpha=1.0)\n\n            x = x.double()\n            labels_a = labels_a.double()\n            labels_b = labels_b.double()\n\n            # take care of the output of the bayesian model and its probabilistic loss\n            if self.bayesian:\n                self.model.double()\n                logit_dict = self.model(x)\n\n                loss = lam * criterion(logit_dict, labels_a)['total_loss'] + (1 - lam) * criterion(\n                    logit_dict, labels_b)['total_loss']\n                #loss = losses_dict['total_loss']\n                logit = criterion(logit_dict, labels_a)['prediction']\n                logit = logit.mean(dim=2)\n            else:\n                self.model.double()\n                logit = self.model(x)\n                loss = lam * criterion(logit, labels_a) + (1 - lam) * criterion(\n                    logit, labels_b\n                )\n        else:\n            \n            if self.bayesian:\n                # measure forward pass time\n                #t_start = time.time()\n                self.model.double()\n                logit_dict = self.model(x)\n                #t_end = time.time() - t_start\n                # logger.info(f'forward pass time: {t_end:.2f} s')\n\n                # criterion is the probabilistic loss class\n                #t_s = time.time()\n                losses_dict = criterion(logit_dict, y)\n                #t_e = time.time() - t_s\n                #logger.info(f'loss time: {t_e:.2f} s')\n                \n                loss = losses_dict['total_loss']\n                logit = losses_dict['prediction'] # Shape: torch.Size([10, 10, 64]) --&gt; (batch_size, num_classes, samples)\n                # change the shape of the logit to be (batch_size, num_classes)\n                logit = logit.mean(dim=2)\n            else:\n                self.model.double()\n                logit = self.model(x)\n                loss = criterion(logit, y)\n        \n        # calculate the number of correct predictions per batch for the bayesian model as well here\n        _, preds = logit.topk(self.topk, 1, True, True)\n\n        loss.backward()\n        ''' ToDo: is it necessary to clip the gradient? it was done in mnvi code\n        Maybe they didn't need it but I'm not sure. For the Bayesian case, it is probably needed.\n        '''\n        if self.bayesian:\n            torch.nn.utils.clip_grad_norm_(self.model.parameters(), 0.1, norm_type='inf')\n        \n        optimizer.step()\n        return loss.item(), torch.sum(preds == y.unsqueeze(1)).item(), y.size(0)\n\n    def _train(\n        self, train_loader, memory_loader, optimizer, criterion\n    ):\n        \n        total_loss, correct, num_data = 0.0, 0.0, 0.0\n\n        self.model.train()\n        if memory_loader is not None and train_loader is not None:\n            data_iterator = zip(train_loader, cycle(memory_loader))\n        elif memory_loader is not None:\n            data_iterator = memory_loader\n        elif train_loader is not None:\n            data_iterator = train_loader\n        else:\n            raise NotImplementedError(\"None of dataloder is valid\")\n        \n        for i, data in enumerate(data_iterator):\n            if len(data) == 2:\n                stream_data, mem_data = data\n                x = torch.cat([stream_data[\"image\"], mem_data[\"image\"]])\n                y = torch.cat([stream_data[\"label\"], mem_data[\"label\"]])\n            else:\n                x = data[\"image\"]\n                y = data[\"label\"]\n            # set to double\n            x = x.double().to(self.device)\n            y = y.double().to(self.device)\n\n            # this is equivalent to the step code in the test repo\n            l, c, d = self.update_model(x, y, criterion, optimizer)\n            # Compute the moving averages - equivalent to MovingAverage in the test repo\n            total_loss += l\n            correct += c\n            num_data += d\n\n        if train_loader is not None:\n            n_batches = len(train_loader)\n        else:\n            n_batches = len(memory_loader)\n\n        return total_loss / n_batches, correct / num_data\n</code></pre>\n<p>but I get this error:</p>\n<pre><code class=\"lang-auto\">\noutputs_mean = F.conv2d(\nRuntimeError: Input type (torch.cuda.FloatTensor) and weight type (torch.cuda.DoubleTensor) should be the same\n</code></pre>\n<p>I did some debugging by printing the dtype of the inputs to the first layer. I see that the batch of iteration 5 is actually in float32!<br>\nAm I doing the casting correct? I already deactivated the cutmix augmentation and the error still persists, so it cannot be the reason for it.<br>\nI appreciate your help on both the stability and the casting to float64.</p>",505          "post_number": 1,506          "post_type": 1,507          "posts_count": 2,508          "updated_at": "2023-03-17T15:21:08.118Z",509          "reply_count": 1,510          "reply_to_post_number": null,511          "quote_count": 0,512          "incoming_link_count": 15,513          "reads": 4,514          "readers_count": 3,515          "score": 80.8,516          "yours": false,517          "topic_id": 175169,518          "topic_slug": "casting-to-double-does-not-work-for-one-of-the-batches-in-the-iteration",519          "display_username": "Samster",520          "primary_group_name": null,521          "flair_name": null,522          "flair_url": null,523          "flair_bg_color": null,524          "flair_color": null,525          "flair_group_id": null,526          "badges_granted": [],527          "version": 1,528          "can_edit": false,529          "can_delete": false,530          "can_recover": false,531          "can_see_hidden_post": false,532          "can_wiki": false,533          "read": true,534          "user_title": "",535          "bookmarked": false,536          "actions_summary": [],537          "moderator": false,538          "admin": false,539          "staff": false,540          "user_id": 56371,541          "hidden": false,542          "trust_level": 1,543          "deleted_at": null,544          "user_deleted": false,545          "edit_reason": null,546          "can_view_edit_history": true,547          "wiki": false,548          "post_url": "/t/casting-to-double-does-not-work-for-one-of-the-batches-in-the-iteration/175169/1",549          "can_accept_answer": false,550          "can_unaccept_answer": false,551          "accepted_answer": false,552          "topic_accepted_answer": null,553          "can_vote": false554        },555        {556          "id": 392781,557          "name": "K. Frank",558          "username": "KFrank",559          "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",560          "created_at": "2023-03-18T01:27:45.281Z",561          "cooked": "<p>Hi Sam!</p>\n<aside class=\"quote no-group quote-modified\" data-username=\"papoo13\" data-post=\"1\" data-topic=\"175169\" data-full=\"true\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/letter_avatar_proxy/v4/letter/p/f19dbf/48.png\" class=\"avatar\"> papoo13:</div>\n<blockquote>\n<pre><code class=\"lang-auto\">        if do_cutmix:\n            x, labels_a, labels_b, lam = cutmix_data(x=x, y=y, alpha=1.0)\n\n            x = x.double()\n...\n        else:\n            \n            if self.bayesian:\n                ...\n                self.model.double()\n                logit_dict = self.model(x)\n</code></pre>\n<p>…</p>\n<pre><code class=\"lang-auto\">RuntimeError: Input type (torch.cuda.FloatTensor) and weight type (torch.cuda.DoubleTensor) should be the same\n</code></pre>\n</blockquote>\n</aside>\n<p>In your <code>if do_cutmix:</code> section of code, you cast <code>x</code> to <code>double()</code>, but<br>\nin the associated <code>else:</code> section, you don’t.  Could the <code>else</code> branch be<br>\ncausing the problem?</p>\n<p>Best.</p>\n<p>K. Frank</p>",562          "post_number": 2,563          "post_type": 1,564          "posts_count": 2,565          "updated_at": "2023-03-18T01:27:45.281Z",566          "reply_count": 0,567          "reply_to_post_number": null,568          "quote_count": 1,569          "incoming_link_count": 2,570          "reads": 4,571          "readers_count": 3,572          "score": 10.8,573          "yours": false,574          "topic_id": 175169,575          "topic_slug": "casting-to-double-does-not-work-for-one-of-the-batches-in-the-iteration",576          "display_username": "K. Frank",577          "primary_group_name": null,578          "flair_name": null,579          "flair_url": null,580          "flair_bg_color": null,581          "flair_color": null,582          "flair_group_id": null,583          "badges_granted": [],584          "version": 1,585          "can_edit": false,586          "can_delete": false,587          "can_recover": false,588          "can_see_hidden_post": false,589          "can_wiki": false,590          "read": true,591          "user_title": null,592          "bookmarked": false,593          "actions_summary": [],594          "moderator": false,595          "admin": false,596          "staff": false,597          "user_id": 18088,598          "hidden": false,599          "trust_level": 2,600          "deleted_at": null,601          "user_deleted": false,602          "edit_reason": null,603          "can_view_edit_history": true,604          "wiki": false,605          "post_url": "/t/casting-to-double-does-not-work-for-one-of-the-batches-in-the-iteration/175169/2",606          "can_accept_answer": false,607          "can_unaccept_answer": false,608          "accepted_answer": false,609          "topic_accepted_answer": null610        }611      ],612      "stream": [613        392730,614        392781615      ]616    },617    "timeline_lookup": [618      [619        1,620        953621      ]622    ],623    "suggested_topics": [624      {625        "fancy_title": "TypeError: &lsquo;int&rsquo; object is not callable for claculating training accuracy",626        "id": 212609,627        "title": "TypeError: 'int' object is not callable for claculating training accuracy",628        "slug": "typeerror-int-object-is-not-callable-for-claculating-training-accuracy",629        "posts_count": 2,630        "reply_count": 0,631        "highest_post_number": 2,632        "image_url": null,633        "created_at": "2024-11-06T11:04:54.598Z",634        "last_posted_at": "2024-11-06T17:20:42.372Z",635        "bumped": true,636        "bumped_at": "2024-11-06T17:20:42.372Z",637        "archetype": "regular",638        "unseen": false,639        "pinned": false,640        "unpinned": null,641        "visible": true,642        "closed": false,643        "archived": false,644        "bookmarked": null,645        "liked": null,646        "tags_descriptions": {},647        "like_count": 0,648        "views": 30,649        "category_id": 5,650        "featured_link": null,651        "has_accepted_answer": false,652        "posters": [653          {654            "extras": null,655            "description": "Original Poster",656            "user": {657              "id": 68222,658              "username": "amy2",659              "name": "amy",660              "avatar_template": "/user_avatar/discuss.pytorch.org/amy2/{size}/62675_2.png",661              "trust_level": 1662            }663          },664          {665            "extras": "latest",666            "description": "Most Recent Poster",667            "user": {668              "id": 3534,669              "username": "ptrblck",670              "name": "",671              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",672              "admin": true,673              "moderator": true,674              "trust_level": 2675            }676          }677        ]678      },679      {680        "fancy_title": "ValueError: You should supply an encoding or a list of encodings to this method that includes input_ids, but you provided [&lsquo;pixel_values&rsquo;]",681        "id": 216220,682        "title": "ValueError: You should supply an encoding or a list of encodings to this method that includes input_ids, but you provided ['pixel_values']",683        "slug": "valueerror-you-should-supply-an-encoding-or-a-list-of-encodings-to-this-method-that-includes-input-ids-but-you-provided-pixel-values",684        "posts_count": 4,685        "reply_count": 1,686        "highest_post_number": 4,687        "image_url": null,688        "created_at": "2025-02-04T13:03:09.113Z",689        "last_posted_at": "2025-02-12T12:11:07.163Z",690        "bumped": true,691        "bumped_at": "2025-02-12T12:30:38.800Z",692        "archetype": "regular",693        "unseen": false,694        "pinned": false,695        "unpinned": null,696        "visible": true,697        "closed": false,698        "archived": false,699        "bookmarked": null,700        "liked": null,701        "tags_descriptions": {},702        "like_count": 0,703        "views": 445,704        "category_id": 5,705        "featured_link": null,706        "has_accepted_answer": false,707        "posters": [708          {709            "extras": "latest",710            "description": "Original Poster, Most Recent Poster",711            "user": {712              "id": 82473,713              "username": "milanalimova",714              "name": null,715              "avatar_template": "/user_avatar/discuss.pytorch.org/milanalimova/{size}/75456_2.png",716              "trust_level": 1717            }718          },719          {720            "extras": null,721            "description": "Frequent Poster",722            "user": {723              "id": 3534,724              "username": "ptrblck",725              "name": "",726              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",727              "admin": true,728              "moderator": true,729              "trust_level": 2730            }731          }732        ]733      },734      {735        "fancy_title": "Switching from &ldquo;model.eval()&rdquo; to &ldquo;model.train()&rdquo; severely degrades performance",736        "id": 215519,737        "title": "Switching from \"model.eval()\" to \"model.train()\" severely degrades performance",738        "slug": "switching-from-model-eval-to-model-train-severely-degrades-performance",739        "posts_count": 1,740        "reply_count": 0,741        "highest_post_number": 1,742        "image_url": null,743        "created_at": "2025-01-17T14:54:50.152Z",744        "last_posted_at": "2025-01-17T14:54:50.195Z",745        "bumped": true,746        "bumped_at": "2025-01-20T06:05:12.301Z",747        "archetype": "regular",748        "unseen": false,749        "pinned": false,750        "unpinned": null,751        "visible": true,752        "closed": false,753        "archived": false,754        "bookmarked": null,755        "liked": null,756        "tags_descriptions": {},757        "like_count": 0,758        "views": 120,759        "category_id": 5,760        "featured_link": null,761        "has_accepted_answer": false,762        "posters": [763          {764            "extras": "latest single",765            "description": "Original Poster, Most Recent Poster",766            "user": {767              "id": 76983,768              "username": "cloudydory",769              "name": "",770              "avatar_template": "/letter_avatar_proxy/v4/letter/c/dfb087/{size}.png",771              "trust_level": 1772            }773          }774        ]775      },776      {777        "fancy_title": "F.scaled_dot_product_attention get query @ key",778        "id": 215697,779        "title": "F.scaled_dot_product_attention get query @ key",780        "slug": "f-scaled-dot-product-attention-get-query-key",781        "posts_count": 1,782        "reply_count": 0,783        "highest_post_number": 1,784        "image_url": null,785        "created_at": "2025-01-22T05:03:27.120Z",786        "last_posted_at": "2025-01-22T05:03:27.155Z",787        "bumped": true,788        "bumped_at": "2025-01-22T05:03:27.155Z",789        "archetype": "regular",790        "unseen": false,791        "pinned": false,792        "unpinned": null,793        "visible": true,794        "closed": false,795        "archived": false,796        "bookmarked": null,797        "liked": null,798        "tags_descriptions": {},799        "like_count": 0,800        "views": 109,801        "category_id": 5,802        "featured_link": null,803        "has_accepted_answer": false,804        "posters": [805          {806            "extras": "latest single",807            "description": "Original Poster, Most Recent Poster",808            "user": {809              "id": 82230,810              "username": "b10901187",811              "name": "閎凱 鍾",812              "avatar_template": "/user_avatar/discuss.pytorch.org/b10901187/{size}/75232_2.png",813              "trust_level": 1814            }815          }816        ]817      },818      {819        "fancy_title": "How to pre-train ResNet using my own dataset",820        "id": 219177,821        "title": "How to pre-train ResNet using my own dataset",822        "slug": "how-to-pre-train-resnet-using-my-own-dataset",823        "posts_count": 2,824        "reply_count": 0,825        "highest_post_number": 2,826        "image_url": null,827        "created_at": "2025-04-17T01:27:09.923Z",828        "last_posted_at": "2025-04-17T20:59:49.425Z",829        "bumped": true,830        "bumped_at": "2025-04-17T20:59:49.425Z",831        "archetype": "regular",832        "unseen": false,833        "pinned": false,834        "unpinned": null,835        "visible": true,836        "closed": false,837        "archived": false,838        "bookmarked": null,839        "liked": null,840        "tags_descriptions": {},841        "like_count": 1,842        "views": 81,843        "category_id": 5,844        "featured_link": null,845        "has_accepted_answer": false,846        "posters": [847          {848            "extras": null,849            "description": "Original Poster",850            "user": {851              "id": 82704,852              "username": "2U3_1967",853              "name": null,854              "avatar_template": "/letter_avatar_proxy/v4/letter/2/cab0a1/{size}.png",855              "trust_level": 1856            }857          },858          {859            "extras": "latest",860            "description": "Most Recent Poster",861            "user": {862              "id": 72430,863              "username": "Eduardo_Lawson",864              "name": "Eduardo Lawson da Silva",865              "avatar_template": "/user_avatar/discuss.pytorch.org/eduardo_lawson/{size}/66899_2.png",866              "trust_level": 2867            }868          }869        ]870      }871    ],872    "tags_descriptions": {},873    "fancy_title": "Casting to double() does not work for one of the batches in the iteration",874    "id": 175169,875    "title": "Casting to double() does not work for one of the batches in the iteration",876    "posts_count": 2,877    "created_at": "2023-03-17T15:21:08.021Z",878    "views": 300,879    "reply_count": 0,880    "like_count": 0,881    "last_posted_at": "2023-03-18T01:27:45.281Z",882    "visible": true,883    "closed": false,884    "archived": false,885    "has_summary": false,886    "archetype": "regular",887    "slug": "casting-to-double-does-not-work-for-one-of-the-batches-in-the-iteration",888    "category_id": 5,889    "word_count": 704,890    "deleted_at": null,891    "user_id": 56371,892    "featured_link": null,893    "pinned_globally": false,894    "pinned_at": null,895    "pinned_until": null,896    "image_url": null,897    "slow_mode_seconds": 0,898    "draft": null,899    "draft_key": "topic_175169",900    "draft_sequence": null,901    "unpinned": null,902    "pinned": false,903    "current_post_number": 1,904    "highest_post_number": 2,905    "deleted_by": null,906    "actions_summary": [907      {908        "id": 4,909        "count": 0,910        "hidden": false,911        "can_act": false912      },913      {914        "id": 8,915        "count": 0,916        "hidden": false,917        "can_act": false918      },919      {920        "id": 10,921        "count": 0,922        "hidden": false,923        "can_act": false924      },925      {926        "id": 7,927        "count": 0,928        "hidden": false,929        "can_act": false930      }931    ],932    "chunk_size": 20,933    "bookmarked": false,934    "topic_timer": null,935    "message_bus_last_id": 0,936    "participant_count": 2,937    "show_read_indicator": false,938    "thumbnails": null,939    "slow_mode_enabled_until": null,940    "can_vote": false,941    "vote_count": 0,942    "user_voted": false,943    "discourse_zendesk_plugin_zendesk_id": null,944    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",945    "details": {946      "can_edit": false,947      "notification_level": 1,948      "participants": [949        {950          "id": 18088,951          "username": "KFrank",952          "name": "K. Frank",953          "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",954          "post_count": 1,955          "primary_group_name": null,956          "flair_name": null,957          "flair_url": null,958          "flair_color": null,959          "flair_bg_color": null,960          "flair_group_id": null,961          "trust_level": 2962        },963        {964          "id": 56371,965          "username": "papoo13",966          "name": "Samster",967          "avatar_template": "/letter_avatar_proxy/v4/letter/p/f19dbf/{size}.png",968          "post_count": 1,969          "primary_group_name": null,970          "flair_name": null,971          "flair_url": null,972          "flair_color": null,973          "flair_bg_color": null,974          "flair_group_id": null,975          "trust_level": 1976        }977      ],978      "created_by": {979        "id": 56371,980        "username": "papoo13",981        "name": "Samster",982        "avatar_template": "/letter_avatar_proxy/v4/letter/p/f19dbf/{size}.png"983      },984      "last_poster": {985        "id": 18088,986        "username": "KFrank",987        "name": "K. Frank",988        "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png"989      }990    },991    "bookmarks": []992  },993  {994    "post_stream": {995      "posts": [996        {997          "id": 392095,998          "name": "",999          "username": "takis17",1000          "avatar_template": "/letter_avatar_proxy/v4/letter/t/ba9def/{size}.png",1001          "created_at": "2023-03-14T17:00:43.449Z",1002          "cooked": "<p>Hello everyone,<br>\nHope you are doing well.</p>\n<p>So I have modified the weights of a given pre-trained model.<br>\nPassing though my evaluation dataset, I cannot detect any changed from before I have change the weights and after with the error induced, even when I changed the weights to extreme cases causing some weights to appear as NaN.</p>\n<p>I have stored the corrupted weights back to a specific layer in the model as such</p>\n<pre><code class=\"lang-auto\"># set manipulated weight to conv layer\nwith torch.no_grad():\n    conv.weight.copy_(weight_float32)\nprint(conv.weight)\n</code></pre>\n<p>My question is that:</p>\n<ol>\n<li>Do I need to re-train the model with the corrupted errors and pass the evaluation dataset<br>\nSince the model’s weight are corrupted  - I double-check the parameters before and after</li>\n</ol>",1003          "post_number": 1,1004          "post_type": 1,1005          "posts_count": 7,1006          "updated_at": "2023-03-14T17:00:43.449Z",1007          "reply_count": 0,1008          "reply_to_post_number": null,1009          "quote_count": 0,1010          "incoming_link_count": 20,1011          "reads": 10,1012          "readers_count": 9,1013          "score": 102.0,1014          "yours": false,1015          "topic_id": 174842,1016          "topic_slug": "evaluation-model-after-weight-errors",1017          "display_username": "",1018          "primary_group_name": null,1019          "flair_name": null,1020          "flair_url": null,1021          "flair_bg_color": null,1022          "flair_color": null,1023          "flair_group_id": null,1024          "badges_granted": [],1025          "version": 1,1026          "can_edit": false,1027          "can_delete": false,1028          "can_recover": false,1029          "can_see_hidden_post": false,1030          "can_wiki": false,1031          "read": true,1032          "user_title": "",1033          "bookmarked": false,1034          "actions_summary": [],1035          "moderator": false,1036          "admin": false,1037          "staff": false,1038          "user_id": 63737,1039          "hidden": false,1040          "trust_level": 1,1041          "deleted_at": null,1042          "user_deleted": false,1043          "edit_reason": null,1044          "can_view_edit_history": true,1045          "wiki": false,1046          "post_url": "/t/evaluation-model-after-weight-errors/174842/1",1047          "can_accept_answer": false,1048          "can_unaccept_answer": false,1049          "accepted_answer": false,1050          "topic_accepted_answer": true,1051          "can_vote": false1052        },1053        {1054          "id": 392161,1055          "name": "",1056          "username": "ptrblck",1057          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1058          "created_at": "2023-03-15T00:12:37.510Z",1059          "cooked": "<p>I’m not sure I understand the issue completely, so please correct me if I miss something.<br>\nBased on your description you are manually manipulating the weights (to e.g. invalid values) which is not reflected in the output and thus you think the weights were never updated or are never used.<br>\nI don’t know if these parameters are ever used, but your code should work as seen here:</p>\n<pre><code class=\"lang-python\">conv = nn.Conv2d(1, 3, 3)\nx = torch.randn(1, 1, 24, 24)\nout = conv(x)\nprint(out.abs().sum())\n#tensor(664.0641, grad_fn=&lt;SumBackward0&gt;)\n\nwith torch.no_grad():\n    conv.weight.copy_(float(\"nan\"))\nprint(conv.weight)\n# Parameter containing:\n# tensor([[[[nan, nan, nan],\n#           [nan, nan, nan],\n#           [nan, nan, nan]]],\n\n\n#         [[[nan, nan, nan],\n#           [nan, nan, nan],\n#           [nan, nan, nan]]],\n\n\n#         [[[nan, nan, nan],\n#           [nan, nan, nan],\n#           [nan, nan, nan]]]], requires_grad=True)\n\nout = conv(x)\nprint(out.abs().sum())\n# tensor(nan, grad_fn=&lt;SumBackward0&gt;)\n</code></pre>",1060          "post_number": 2,1061          "post_type": 1,1062          "posts_count": 7,1063          "updated_at": "2023-03-15T00:12:37.510Z",1064          "reply_count": 1,1065          "reply_to_post_number": null,1066          "quote_count": 0,1067          "incoming_link_count": 1,1068          "reads": 10,1069          "readers_count": 9,1070          "score": 12.0,1071          "yours": false,1072          "topic_id": 174842,1073          "topic_slug": "evaluation-model-after-weight-errors",1074          "display_username": "",1075          "primary_group_name": null,1076          "flair_name": null,1077          "flair_url": null,1078          "flair_bg_color": null,1079          "flair_color": null,1080          "flair_group_id": null,1081          "badges_granted": [],1082          "version": 1,1083          "can_edit": false,1084          "can_delete": false,1085          "can_recover": false,1086          "can_see_hidden_post": false,1087          "can_wiki": false,1088          "read": true,1089          "user_title": "",1090          "bookmarked": false,1091          "actions_summary": [],1092          "moderator": true,1093          "admin": true,1094          "staff": true,1095          "user_id": 3534,1096          "hidden": false,1097          "trust_level": 2,1098          "deleted_at": null,1099          "user_deleted": false,1100          "edit_reason": null,1101          "can_view_edit_history": true,1102          "wiki": false,1103          "post_url": "/t/evaluation-model-after-weight-errors/174842/2",1104          "can_accept_answer": false,1105          "can_unaccept_answer": false,1106          "accepted_answer": false,1107          "topic_accepted_answer": true1108        },1109        {1110          "id": 392309,1111          "name": "",1112          "username": "takis17",1113          "avatar_template": "/letter_avatar_proxy/v4/letter/t/ba9def/{size}.png",1114          "created_at": "2023-03-15T19:59:46.692Z",1115          "cooked": "<p><a class=\"mention\" href=\"/u/ptrblck\">@ptrblck</a></p>\n<p>Thank you, really appreciate your help.<br>\nCorrect, that is what I am doing, except for the only difference that not all the values are nan - I randomly shuffle values and change them to one.<br>\nMy code follows the same logic as the one you provided, but I wanted to verify that I am actually correctly copying the weights back to my model - since at the end a tensor is immutable, that is why I converted it to a numpy array.<br>\nIf that is true ; then<br>\nDoes it make sense re-training the model and then pass the evaluation dataset (I imagine by doing that the weights will adapt but will still be different from the “original” for example if I train it for 1 epoch) hence might detect some changes<br>\nOr just pass it through after model.eval()</p>",1116          "post_number": 3,1117          "post_type": 1,1118          "posts_count": 7,1119          "updated_at": "2023-03-15T19:59:46.692Z",1120          "reply_count": 1,1121          "reply_to_post_number": 2,1122          "quote_count": 0,1123          "incoming_link_count": 0,1124          "reads": 6,1125          "readers_count": 5,1126          "score": 6.2,1127          "yours": false,1128          "topic_id": 174842,1129          "topic_slug": "evaluation-model-after-weight-errors",1130          "display_username": "",1131          "primary_group_name": null,1132          "flair_name": null,1133          "flair_url": null,1134          "flair_bg_color": null,1135          "flair_color": null,1136          "flair_group_id": null,1137          "badges_granted": [],1138          "version": 1,1139          "can_edit": false,1140          "can_delete": false,1141          "can_recover": false,1142          "can_see_hidden_post": false,1143          "can_wiki": false,1144          "read": true,1145          "user_title": "",1146          "reply_to_user": {1147            "id": 3534,1148            "username": "ptrblck",1149            "name": "",1150            "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"1151          },1152          "bookmarked": false,1153          "actions_summary": [],1154          "moderator": false,1155          "admin": false,1156          "staff": false,1157          "user_id": 63737,1158          "hidden": false,1159          "trust_level": 1,1160          "deleted_at": null,1161          "user_deleted": false,1162          "edit_reason": null,1163          "can_view_edit_history": true,1164          "wiki": false,1165          "post_url": "/t/evaluation-model-after-weight-errors/174842/3",1166          "can_accept_answer": false,1167          "can_unaccept_answer": false,1168          "accepted_answer": false,1169          "topic_accepted_answer": true1170        },1171        {1172          "id": 392329,1173          "name": "",1174          "username": "ptrblck",1175          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1176          "created_at": "2023-03-16T00:31:18.008Z",1177          "cooked": "<p>I think it depends on your actual use case if a re-training is desired or not.<br>\nThe main issue I saw from your original post was that it seemed as if the weight changes were not used at all, which should not be the case as seen in my code snippet. Also setting a single value would already reflect the changes without any re-training:</p>\n<pre><code class=\"lang-python\">conv = nn.Conv2d(1, 3, 3)\nx = torch.randn(1, 1, 24, 24)\nout = conv(x)\nprint(out.abs().sum())\n#tensor(664.0641, grad_fn=&lt;SumBackward0&gt;)\n\nwith torch.no_grad():\n    conv.weight[0, 0, 0, 0].copy_(float(\"nan\"))\nprint(conv.weight)\n# Parameter containing:\n# tensor([[[[    nan,  0.0670,  0.3143],\n#           [-0.3256, -0.1418, -0.2889],\n#           [ 0.2058, -0.0386, -0.0661]]],\n\n\n#         [[[ 0.1033, -0.0968, -0.1645],\n#           [-0.2478,  0.0022, -0.1859],\n#           [ 0.1836, -0.2671,  0.1425]]],\n\n\n#         [[[-0.1508, -0.3039, -0.3118],\n#           [ 0.2929, -0.2692, -0.1859],\n#           [ 0.1304, -0.0377,  0.0886]]]], requires_grad=True)\n\nout = conv(x)\nprint(out.abs().sum())\n# tensor(nan, grad_fn=&lt;SumBackward0&gt;)\n</code></pre>\n<p>I don’t know where the new values come from, but assuming you are trying to use other pre-trained weights a re-training might not be necessary.</p>",1178          "post_number": 4,1179          "post_type": 1,1180          "posts_count": 7,1181          "updated_at": "2023-03-16T00:31:18.008Z",1182          "reply_count": 1,1183          "reply_to_post_number": 3,1184          "quote_count": 0,1185          "incoming_link_count": 0,1186          "reads": 6,1187          "readers_count": 5,1188          "score": 6.2,1189          "yours": false,1190          "topic_id": 174842,1191          "topic_slug": "evaluation-model-after-weight-errors",1192          "display_username": "",1193          "primary_group_name": null,1194          "flair_name": null,1195          "flair_url": null,1196          "flair_bg_color": null,1197          "flair_color": null,1198          "flair_group_id": null,1199          "badges_granted": [],1200          "version": 1,

Showing the first 1,200 of 61985 lines. Download the file for the rest.