CoolFace
Datasetpublic

Anurag1734/cuda-error-resolution-analysis

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes7downloads
topics_batch_9.json65950 linesDownload Raw Back to raw
1[2  {3    "post_stream": {4      "posts": [5        {6          "id": 471672,7          "name": "Natalia ",8          "username": "NataliaZagalabs",9          "avatar_template": "/letter_avatar_proxy/v4/letter/n/85e7bf/{size}.png",10          "created_at": "2025-06-11T15:07:01.464Z",11          "cooked": "<p>Una empresa colombiana en crecimiento global busca un/a <strong>Desarrollador/a con experiencia en Machine Learning</strong>, altamente competente en <strong>Python</strong>, para sumarse a un equipo internacional.<br>\nSi tienes conocimientos sólidos en <strong>PyTorch, Torchaudio y Torchvideo</strong>, te apasiona la inteligencia artificial aplicada a audio y video, y hablas inglés con fluidez, ¡esta oportunidad es para ti!<br>\n<img src=\"https://discuss.pytorch.org/images/emoji/apple/round_pushpin.png?v=14\" title=\":round_pushpin:\" class=\"emoji\" alt=\":round_pushpin:\" loading=\"lazy\" width=\"20\" height=\"20\"> Modalidad: Trabajo <strong>100% remoto</strong></p>\n<ul>\n<li>Contratación desde <strong>Colombia o Argentina</strong></li>\n<li>Pagos en <strong>USD</strong></li>\n</ul>\n<p>CV <a href=\"mailto:natalia.ramirez@zagalabs.com\">natalia.ramirez@zagalabs.com</a></p>",12          "post_number": 1,13          "post_type": 1,14          "posts_count": 2,15          "updated_at": "2025-06-11T15:15:42.656Z",16          "reply_count": 0,17          "reply_to_post_number": null,18          "quote_count": 0,19          "incoming_link_count": 3,20          "reads": 18,21          "readers_count": 17,22          "score": 18.6,23          "yours": false,24          "topic_id": 220733,25          "topic_slug": "oferta-laboral-payton-con-ml-pytorch-torchaudio-torchvideo",26          "display_username": "Natalia ",27          "primary_group_name": null,28          "flair_name": null,29          "flair_url": null,30          "flair_bg_color": null,31          "flair_color": null,32          "flair_group_id": null,33          "badges_granted": [],34          "version": 2,35          "can_edit": false,36          "can_delete": false,37          "can_recover": false,38          "can_see_hidden_post": false,39          "can_wiki": false,40          "read": true,41          "user_title": null,42          "bookmarked": false,43          "actions_summary": [],44          "moderator": false,45          "admin": false,46          "staff": false,47          "user_id": 84655,48          "hidden": false,49          "trust_level": 0,50          "deleted_at": null,51          "user_deleted": false,52          "edit_reason": null,53          "can_view_edit_history": true,54          "wiki": false,55          "post_url": "/t/oferta-laboral-payton-con-ml-pytorch-torchaudio-torchvideo/220733/1",56          "can_accept_answer": false,57          "can_unaccept_answer": false,58          "accepted_answer": false,59          "topic_accepted_answer": null,60          "can_vote": false61        },62        {63          "id": 471673,64          "name": "",65          "username": "ptrblck",66          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",67          "created_at": "2025-06-11T15:15:36.963Z",68          "cooked": "<p>From Google Translate:</p>\n<pre><code class=\"lang-auto\">A Colombian company with global growth is seeking a Machine Learning Developer with extensive Python experience to join an international team.\nIf you have solid knowledge of PyTorch, Torchaudio, and Torchvideo, are passionate about artificial intelligence applied to audio and video, and are fluent in English, this opportunity is for you!\n:round_pushpin: Working Mode: 100% remote\n\nHiring from Colombia or Argentina\nPayments in USD```</code></pre>",69          "post_number": 2,70          "post_type": 1,71          "posts_count": 2,72          "updated_at": "2025-06-11T15:15:36.963Z",73          "reply_count": 0,74          "reply_to_post_number": null,75          "quote_count": 0,76          "incoming_link_count": 2,77          "reads": 18,78          "readers_count": 17,79          "score": 28.6,80          "yours": false,81          "topic_id": 220733,82          "topic_slug": "oferta-laboral-payton-con-ml-pytorch-torchaudio-torchvideo",83          "display_username": "",84          "primary_group_name": null,85          "flair_name": null,86          "flair_url": null,87          "flair_bg_color": null,88          "flair_color": null,89          "flair_group_id": null,90          "badges_granted": [],91          "version": 1,92          "can_edit": false,93          "can_delete": false,94          "can_recover": false,95          "can_see_hidden_post": false,96          "can_wiki": false,97          "read": true,98          "user_title": "",99          "bookmarked": false,100          "actions_summary": [101            {102              "id": 2,103              "count": 1104            }105          ],106          "moderator": true,107          "admin": true,108          "staff": true,109          "user_id": 3534,110          "hidden": false,111          "trust_level": 2,112          "deleted_at": null,113          "user_deleted": false,114          "edit_reason": null,115          "can_view_edit_history": true,116          "wiki": false,117          "post_url": "/t/oferta-laboral-payton-con-ml-pytorch-torchaudio-torchvideo/220733/2",118          "can_accept_answer": false,119          "can_unaccept_answer": false,120          "accepted_answer": false,121          "topic_accepted_answer": null122        }123      ],124      "stream": [125        471672,126        471673127      ]128    },129    "timeline_lookup": [130      [131        1,132        136133      ]134    ],135    "suggested_topics": [136      {137        "fancy_title": "We&rsquo;re Hiring: ML Engineer (Audio Signal Processing) :musical_notes:",138        "id": 218871,139        "title": "We're Hiring: ML Engineer (Audio Signal Processing) 🎶",140        "slug": "were-hiring-ml-engineer-audio-signal-processing",141        "posts_count": 1,142        "reply_count": 0,143        "highest_post_number": 1,144        "image_url": null,145        "created_at": "2025-04-08T14:13:21.488Z",146        "last_posted_at": "2025-04-08T14:13:21.528Z",147        "bumped": true,148        "bumped_at": "2025-04-08T14:13:21.528Z",149        "archetype": "regular",150        "unseen": false,151        "pinned": false,152        "unpinned": null,153        "visible": true,154        "closed": false,155        "archived": false,156        "bookmarked": null,157        "liked": null,158        "tags_descriptions": {},159        "like_count": 0,160        "views": 233,161        "category_id": 24,162        "featured_link": null,163        "has_accepted_answer": false,164        "posters": [165          {166            "extras": "latest single",167            "description": "Original Poster, Most Recent Poster",168            "user": {169              "id": 83680,170              "username": "durai",171              "name": "",172              "avatar_template": "/letter_avatar_proxy/v4/letter/d/34f0e0/{size}.png",173              "trust_level": 0174            }175          }176        ]177      },178      {179        "fancy_title": "Looking for Pytorch Developer",180        "id": 216814,181        "title": "Looking for Pytorch Developer",182        "slug": "looking-for-pytorch-developer",183        "posts_count": 3,184        "reply_count": 1,185        "highest_post_number": 3,186        "image_url": null,187        "created_at": "2025-02-18T08:12:40.307Z",188        "last_posted_at": "2025-02-20T10:55:23.972Z",189        "bumped": true,190        "bumped_at": "2025-02-20T10:55:23.972Z",191        "archetype": "regular",192        "unseen": false,193        "pinned": false,194        "unpinned": null,195        "visible": true,196        "closed": false,197        "archived": false,198        "bookmarked": null,199        "liked": null,200        "tags_descriptions": {},201        "like_count": 0,202        "views": 254,203        "category_id": 24,204        "featured_link": null,205        "has_accepted_answer": false,206        "posters": [207          {208            "extras": null,209            "description": "Original Poster",210            "user": {211              "id": 82757,212              "username": "Aglowid_IT_Solutions",213              "name": "Aglowid IT Solutions",214              "avatar_template": "/user_avatar/discuss.pytorch.org/aglowid_it_solutions/{size}/75716_2.png",215              "trust_level": 0216            }217          },218          {219            "extras": "latest",220            "description": "Most Recent Poster",221            "user": {222              "id": 82779,223              "username": "Jay_G",224              "name": "Jay G",225              "avatar_template": "/user_avatar/discuss.pytorch.org/jay_g/{size}/75740_2.png",226              "trust_level": 0227            }228          }229        ]230      },231      {232        "fancy_title": "Hiring: ML Engineers @ Red Hat",233        "id": 217114,234        "title": "Hiring: ML Engineers @ Red Hat",235        "slug": "hiring-ml-engineers-red-hat",236        "posts_count": 1,237        "reply_count": 0,238        "highest_post_number": 1,239        "image_url": null,240        "created_at": "2025-02-24T20:13:02.436Z",241        "last_posted_at": "2025-02-24T20:13:02.482Z",242        "bumped": true,243        "bumped_at": "2025-02-25T14:45:06.281Z",244        "archetype": "regular",245        "unseen": false,246        "pinned": false,247        "unpinned": null,248        "visible": true,249        "closed": false,250        "archived": false,251        "bookmarked": null,252        "liked": null,253        "tags_descriptions": {},254        "like_count": 0,255        "views": 300,256        "category_id": 24,257        "featured_link": null,258        "has_accepted_answer": false,259        "posters": [260          {261            "extras": "latest single",262            "description": "Original Poster, Most Recent Poster",263            "user": {264              "id": 82907,265              "username": "mejones_redhat",266              "name": "",267              "avatar_template": "/letter_avatar_proxy/v4/letter/m/77aa72/{size}.png",268              "trust_level": 0269            }270          }271        ]272      },273      {274        "fancy_title": "Loading and retraining Peft model error ( grad and does not have a grad_fn)",275        "id": 216684,276        "title": "Loading and retraining Peft model error ( grad and does not have a grad_fn)",277        "slug": "loading-and-retraining-peft-model-error-grad-and-does-not-have-a-grad-fn",278        "posts_count": 1,279        "reply_count": 0,280        "highest_post_number": 1,281        "image_url": null,282        "created_at": "2025-02-14T17:01:40.154Z",283        "last_posted_at": "2025-02-14T17:01:40.194Z",284        "bumped": true,285        "bumped_at": "2025-02-15T02:11:43.680Z",286        "archetype": "regular",287        "unseen": false,288        "pinned": false,289        "unpinned": null,290        "visible": true,291        "closed": false,292        "archived": false,293        "bookmarked": null,294        "liked": null,295        "tags_descriptions": {},296        "like_count": 0,297        "views": 128,298        "category_id": 1,299        "featured_link": null,300        "has_accepted_answer": false,301        "posters": [302          {303            "extras": "latest single",304            "description": "Original Poster, Most Recent Poster",305            "user": {306              "id": 82364,307              "username": "Sourabh_Yadav",308              "name": "Sourabh Yadav",309              "avatar_template": "/user_avatar/discuss.pytorch.org/sourabh_yadav/{size}/75350_2.png",310              "trust_level": 1311            }312          }313        ]314      },315      {316        "fancy_title": "CNN predicts constant values for sparse amplitude regression — can&rsquo;t learn true pixel values",317        "id": 220387,318        "title": "CNN predicts constant values for sparse amplitude regression — can't learn true pixel values",319        "slug": "cnn-predicts-constant-values-for-sparse-amplitude-regression-cant-learn-true-pixel-values",320        "posts_count": 10,321        "reply_count": 7,322        "highest_post_number": 10,323        "image_url": "https://discuss.pytorch.org/uploads/default/optimized/3X/a/3/a34222a0d2908564d4addbd58d278604e98dbfaa_2_1024x528.png",324        "created_at": "2025-05-27T22:23:28.017Z",325        "last_posted_at": "2025-05-30T12:25:47.403Z",326        "bumped": true,327        "bumped_at": "2025-05-30T12:25:47.403Z",328        "archetype": "regular",329        "unseen": false,330        "pinned": false,331        "unpinned": null,332        "visible": true,333        "closed": false,334        "archived": false,335        "bookmarked": null,336        "liked": null,337        "tags_descriptions": {},338        "like_count": 0,339        "views": 105,340        "category_id": 1,341        "featured_link": null,342        "has_accepted_answer": false,343        "posters": [344          {345            "extras": "latest",346            "description": "Original Poster, Most Recent Poster",347            "user": {348              "id": 84482,349              "username": "thunlok",350              "name": "",351              "avatar_template": "/letter_avatar_proxy/v4/letter/t/c4cdca/{size}.png",352              "trust_level": 0353            }354          },355          {356            "extras": null,357            "description": "Frequent Poster",358            "user": {359              "id": 18088,360              "username": "KFrank",361              "name": "K. Frank",362              "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",363              "trust_level": 2364            }365          }366        ]367      }368    ],369    "tags_descriptions": {},370    "fancy_title": "Oferta Laboral Payton con ML, Pytorch, Torchaudio, Torchvideo",371    "id": 220733,372    "title": "Oferta Laboral Payton con ML, Pytorch, Torchaudio, Torchvideo",373    "posts_count": 2,374    "created_at": "2025-06-11T15:07:01.426Z",375    "views": 108,376    "reply_count": 0,377    "like_count": 1,378    "last_posted_at": "2025-06-11T15:15:36.963Z",379    "visible": true,380    "closed": false,381    "archived": false,382    "has_summary": false,383    "archetype": "regular",384    "slug": "oferta-laboral-payton-con-ml-pytorch-torchaudio-torchvideo",385    "category_id": 24,386    "word_count": 139,387    "deleted_at": null,388    "user_id": 84655,389    "featured_link": null,390    "pinned_globally": false,391    "pinned_at": null,392    "pinned_until": null,393    "image_url": null,394    "slow_mode_seconds": 0,395    "draft": null,396    "draft_key": "topic_220733",397    "draft_sequence": null,398    "unpinned": null,399    "pinned": false,400    "current_post_number": 1,401    "highest_post_number": 2,402    "deleted_by": null,403    "actions_summary": [404      {405        "id": 4,406        "count": 0,407        "hidden": false,408        "can_act": false409      },410      {411        "id": 8,412        "count": 0,413        "hidden": false,414        "can_act": false415      },416      {417        "id": 10,418        "count": 0,419        "hidden": false,420        "can_act": false421      },422      {423        "id": 7,424        "count": 0,425        "hidden": false,426        "can_act": false427      }428    ],429    "chunk_size": 20,430    "bookmarked": false,431    "topic_timer": null,432    "message_bus_last_id": 0,433    "participant_count": 2,434    "show_read_indicator": false,435    "thumbnails": null,436    "slow_mode_enabled_until": null,437    "can_vote": false,438    "vote_count": 0,439    "user_voted": false,440    "discourse_zendesk_plugin_zendesk_id": null,441    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",442    "details": {443      "can_edit": false,444      "notification_level": 1,445      "participants": [446        {447          "id": 3534,448          "username": "ptrblck",449          "name": "",450          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",451          "post_count": 1,452          "primary_group_name": null,453          "flair_name": null,454          "flair_url": null,455          "flair_color": null,456          "flair_bg_color": null,457          "flair_group_id": null,458          "admin": true,459          "moderator": true,460          "trust_level": 2461        },462        {463          "id": 84655,464          "username": "NataliaZagalabs",465          "name": "Natalia ",466          "avatar_template": "/letter_avatar_proxy/v4/letter/n/85e7bf/{size}.png",467          "post_count": 1,468          "primary_group_name": null,469          "flair_name": null,470          "flair_url": null,471          "flair_color": null,472          "flair_bg_color": null,473          "flair_group_id": null,474          "trust_level": 0475        }476      ],477      "created_by": {478        "id": 84655,479        "username": "NataliaZagalabs",480        "name": "Natalia ",481        "avatar_template": "/letter_avatar_proxy/v4/letter/n/85e7bf/{size}.png"482      },483      "last_poster": {484        "id": 3534,485        "username": "ptrblck",486        "name": "",487        "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"488      }489    },490    "bookmarks": []491  },492  {493    "post_stream": {494      "posts": [495        {496          "id": 471661,497          "name": "Jie Huang",498          "username": "shangxiaaabb",499          "avatar_template": "/user_avatar/discuss.pytorch.org/shangxiaaabb/{size}/76694_2.png",500          "created_at": "2025-06-11T14:11:37.246Z",501          "cooked": "<p>I ran the same code on both A100 and 4090 GPUs, but encountered an issue only on the A100.</p>\n<p><strong>Problem Description:</strong><br>\nOn the A100, GPU memory usage on the main device increases gradually with each epoch. For example, at <code>epoch=1</code>, CUDA memory usage is around <strong>75GB</strong>, but by <code>epoch=4</code>, it grows to <strong>80GB</strong>, eventually leading to an <strong>Out Of Memory (OOM)</strong> error.<br>\nHowever, the same code running on the <strong>4090</strong> does <strong>not</strong> exhibit this issue — GPU memory remains stable throughout training.<br>\nA100:CUDA Version: 12.9,Name: accelerate  Version: 1.7.0,Name: torch  Version: 2.7.0+cu128;<br>\n4090:CUDA Version: 12.4,Name: accelerate  Version: 1.3.0;Name: torch  Version: 2.5.1;<br>\nMy Code like that:</p>\n<pre data-code-wrap=\"python\"><code class=\"lang-python\">import sys\nimport os\nimport warnings\nsys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), '../')))\nsys.path.append(os.path.abspath(os.path.join(os.path.dirname(__file__), './')))\nos.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\nos.environ['CURL_CA_BUNDLE'] = ''\nwarnings.filterwarnings(\"ignore\")\n\nfrom tqdm import tqdm\nimport numpy as np\nimport torch\nfrom torch.utils.data import DataLoader\nfrom accelerate import Accelerator\nfrom accelerate.utils import DistributedDataParallelKwargs\nfrom sklearn.metrics import accuracy_score, f1_score, precision_score, recall_score, classification_report\n\nfrom config import MyConfig\nfrom model.model import MyModel\nfrom data_loader import MyDataset\nfrom evaluate import compute_acc_f1_recall\n\nconfig = MyConfig()\n\ndef train(epoch, model, optimizer, loss_function, data_loader, accelerator, lr_scheduler):\n    model.train()\n    progress_bar = tqdm(total=len(data_loader), \n                            disable=not accelerator.is_main_process, \n                            desc=f\"Epoch-TRAIN {epoch}\")\n\n    for i, batch in enumerate(data_loader):\n        with accelerator.accumulate(model):\n            image_data, label_data, text_data, bbox_data, text_padding, bbox_padding = batch\n            label_data = label_data.to(accelerator.device)\n            image_data = image_data.to(accelerator.device)\n            bbox_data = bbox_data.to(accelerator.device)\n\n            if config.bbox_embedding:\n                bbox_padding = bbox_padding.to(accelerator.device)\n                out = model(image_data, bbox_data, text_data, bbox_padding)\n            else:\n                out = model(image_data, bbox_data, text_data)\n\n            bbox_padding_mask = bbox_padding.to(dtype=torch.bool, device=accelerator.device)\n            valid_mask = ~bbox_padding_mask\n            valid_out = out[valid_mask]\n            valid_labels = label_data[valid_mask]\n            valid_labels = torch.argmax(valid_labels, dim=-1)\n\n            loss = loss_function(valid_out, valid_labels)\n\n            accelerator.backward(loss)\n            if accelerator.sync_gradients:\n                accelerator.clip_grad_norm_(model.parameters(), 1.0)\n            optimizer.step()\n            lr_scheduler.step()\n            optimizer.zero_grad()\n\n            if accelerator.is_main_process:\n                acc, f1, recall, balanced_acc = compute_acc_f1_recall(preds= valid_out, labels= valid_labels)\n        progress_bar.update(1)\n        if accelerator.is_main_process:\n            logs = {\"Train/loss\": loss.item(), \n                    \"Train/lr\": lr_scheduler.get_last_lr()[0],\n                    \"Train/ACC\": acc, \n                    \"Train/F1-Micro\": f1, \n                    \"Train/Recall\": recall, \n                    \"Train/ACC-Balance\": balanced_acc}\n            progress_bar.set_postfix(\n                loss=loss.item(), lr=lr_scheduler.get_last_lr()[0],\n                acc=acc, f1=f1)\n            accelerator.log(logs)\n\n\ndef test(epoch, model, loss_function, data_loader, accelerator):\n    model.eval()\n    progress_bar = tqdm(total=len(data_loader),\n                        disable=not accelerator.is_main_process, \n                        desc=f\"Epoch-TEST {epoch}\")\n\n    mean_acc, mean_f1 = 0.0, 0.0\n    with torch.no_grad():\n        for i, batch in enumerate(data_loader):\n            image_data, label_data, text_data, bbox_data, text_padding, bbox_padding = batch\n            label_data = label_data.to(accelerator.device)\n            image_data = image_data.to(accelerator.device)\n            bbox_data = bbox_data.to(accelerator.device)\n\n            if config.bbox_embedding:\n                bbox_padding = bbox_padding.to(accelerator.device)\n                out = model(image_data, bbox_data, text_data, bbox_padding)\n            else:\n                out = model(image_data, bbox_data, text_data)\n\n            bbox_padding_mask = bbox_padding.to(dtype=torch.bool, device=accelerator.device)\n            valid_mask = ~bbox_padding_mask \n            valid_out = out[valid_mask] \n            valid_labels = label_data[valid_mask]  \n            valid_labels = torch.argmax(valid_labels, dim=-1)\n\n            loss = loss_function(valid_out, valid_labels)\n            acc, f1, recall, balanced_acc = compute_acc_f1_recall(preds= valid_out, labels= valid_labels)\n\n            progress_bar.update(1)\n            logs = {\"Test/loss\": loss.item(), \n                    \"Test/ACC\": acc, \n                    \"Test/F1-Micro\": f1, \n                    \"Test/Recall\": recall, \n                    \"Test/ACC-Balance\": balanced_acc}\n            progress_bar.set_postfix(\n                loss=loss.item(),\n                acc=acc, f1=f1)\n            mean_acc += acc\n            mean_f1 += f1\n            accelerator.log(logs)\n    return mean_acc/len(data_loader), mean_f1/ len(data_loader)\n\n\n@torch.no_grad\ndef evaluate(model, pth_path, data_loader, accelerator, data_type: str='test'):\n    if config.data_type== 'latex':\n        labels =....\n        index_to_label = {idx: label for idx, label in enumerate(labels)}\n    all_preds, all_labels = [], []\n    if pth_path is not None:\n        checkpoint = torch.load(pth_path, map_location=accelerator.device, weights_only= False)\n        try:\n            model.module.load_state_dict(checkpoint['model_state_dict'])\n        except Exception:\n            model.load_state_dict(checkpoint['model_state_dict'])\n    model.eval()\n\n    with tqdm(total= len(data_loader), desc=f'Eva-{data_type}') as pbar:\n        for i, batch in enumerate(data_loader):\n            image_data, label_data, text_data, bbox_data, text_padding, bbox_padding = batch\n            label_data = label_data.to(accelerator.device)\n            image_data = image_data.to(accelerator.device)\n            bbox_data = bbox_data.to(accelerator.device)\n\n            if config.bbox_embedding:\n                bbox_padding = bbox_padding.to(accelerator.device)\n                out = model(image_data, bbox_data, text_data, bbox_padding)\n            else:\n                out = model(image_data, bbox_data, text_data)\n\n            bbox_padding_mask = bbox_padding.to(dtype=torch.bool, device=accelerator.device)\n            valid_mask = ~bbox_padding_mask \n            valid_out = out[valid_mask]\n            valid_labels = label_data[valid_mask]\n            valid_labels = torch.argmax(valid_labels, dim=-1)\n\n            preds = torch.argmax(valid_out, dim=-1) \n                \n            all_preds.append(preds.detach().cpu().numpy())\n            all_labels.append(valid_labels.detach().cpu().numpy())\n            pbar.update(1)\n    ......\n\ndef main(image_model_name= None, features_fushion=None):\n    if image_model_name:\n        config.image_model_name = image_model_name\n        config.features_fushion = features_fushion\n        config.output_dir = f\"{config.output_dir}/{config.features_fushion}\"\n\n    kwargs_handlers=[DistributedDataParallelKwargs(find_unused_parameters=False)]\n    log_writing = \"tensorboard\" # if config.small_dataset else [\"tensorboard\", \"wandb\"]\n    accelerator = Accelerator(mixed_precision= config.mixed_precision, \n                              gradient_accumulation_steps= config.gradient_accumulation_steps,\n                              log_with= log_writing,\n                              project_dir=os.path.join(config.output_dir, f\"logs\"),\n                              kwargs_handlers= kwargs_handlers\n                              )\n    if accelerator.is_main_process:\n        os.makedirs(config.output_dir, exist_ok=True)\n        accelerator.init_trackers(f\"Train-{config.pred_heads}\")\n     \n    # data\n    train_dataset = MyDataset(...)\n    test_dataset = MyDataset(...)\n    train_dataloader = DataLoader(train_dataset, batch_size= config.batch_size, \n                                  collate_fn= train_dataset.collate_fn, num_workers= 4)\n    test_dataloader = DataLoader(test_dataset, batch_size= config.batch_size,\n                                 collate_fn= test_dataset.collate_fn)\n\n    # model\n    model = MyModel(...)\n    \n    if config.lora:\n        optimizer = torch.optim.AdamW([\n            {'params': model.image_model.parameters(), 'lr': 2e-4, 'weight_decay': 1e-2},\n            {'params': model.text_model.parameters(), 'lr': 4e-5, 'weight_decay': 1e-2},\n            {'params': [p for n, p in model.named_parameters() \n                if 'image_model' not in n and 'text_model' not in n]},\n        ], lr= config.learning_rate)\n    else:\n        optimizer = torch.optim.AdamW(model.parameters(), lr= config.learning_rate)\n    lr_scheduler = torch.optim.lr_scheduler.LinearLR(optimizer,\n                                                     start_factor=0.1,\n                                                     total_iters= 10 * len(train_dataloader))\n    loss_function = torch.nn.CrossEntropyLoss()\n\n    model, optimizer, train_dataloader, test_dataloader, lr_scheduler = accelerator.prepare(model, optimizer, \n                                                                                            train_dataloader, \n                                                                                            test_dataloader, \n                                                                                            lr_scheduler)\n    \n    best_acc, best_f1 = 0.0, 0.0\n    for epoch in range(config.epochs):\n        train(epoch, model, optimizer, loss_function, train_dataloader, \n              accelerator, lr_scheduler)\n        if accelerator.is_main_process:\n            mean_acc, mean_f1 = test(epoch, model,loss_function, test_dataloader, accelerator)\n            if mean_acc&gt;= best_acc+ 0.01:\n                best_acc = mean_acc\n                # model_to_save = accelerator.unwrap_model(model)\n                model_to_save = accelerator.get_state_dict(model)\n                accelerator.save({\n                    'model_state_dict': model_to_save,\n                    'optimizer_state_dict': optimizer.state_dict(),\n                    'lr_scheduler_state_dict': lr_scheduler.state_dict(),\n                    'epoch': epoch,\n                    'best_acc': best_acc,\n                    'config': vars(config)\n                }, f\"{config.output_dir}/model_acc_best.pth\")\n                torch.cuda.empty_cache()\n            if mean_f1&gt;= best_f1+ 0.01:\n                best_f1 = mean_f1\n                # model_to_save = accelerator.unwrap_model(model)\n                model_to_save = accelerator.get_state_dict(model)\n                accelerator.save({\n                    'model_state_dict': model_to_save,\n                    'optimizer_state_dict': optimizer.state_dict(),\n                    'lr_scheduler_state_dict': lr_scheduler.state_dict(),\n                    'epoch': epoch,\n                    'best_f1': best_f1,\n                    'config': vars(config)\n                }, f\"{config.output_dir}/model_f1_best.pth\")\n                torch.cuda.empty_cache()\n\n            if epoch% 5== 0 or epoch== config.epochs- 1:\n                for _ in [(\"TEST-ACC\", f\"{config.output_dir}/model_acc_best.pth\"), (\"TEST-F1\", f\"{config.output_dir}/model_f1_best.pth\")]:\n                    checkpoint = torch.load(_[1], map_location=accelerator.device, weights_only= False)\n                    try:\n                        model.module.load_state_dict(checkpoint['model_state_dict'])\n                    except Exception:\n                        model.load_state_dict(checkpoint['model_state_dict'])\n                    evaluate(model, None, data_loader= test_dataloader, accelerator=accelerator,\n                             data_type= _[0])\n                    del checkpoint\n                    torch.cuda.empty_cache()\n        torch.cuda.empty_cache()\n    \n    accelerator.end_training()\n\nif __name__ == '__main__':\n    # CUDA_VISIBLE_DEVICES=1,2,3 accelerate launch --num_processes=3 train.py\n    # CUDA_VISIBLE_DEVICES=2,3 accelerate launch --num_processes=2 train.py\n    import argparse\n    parser = argparse.ArgumentParser(description=\"Run main function with parameters\")\n    parser.add_argument('--image_model_name', type=str, default=None, help='Name of the image model')\n    parser.add_argument('--features_fusion', type=str, default=None, help='Type of features fusion')\n    args = parser.parse_args()\n    main(args.image_model_name, args.features_fusion)\n</code></pre>",502          "post_number": 1,503          "post_type": 1,504          "posts_count": 3,505          "updated_at": "2025-06-11T14:14:53.257Z",506          "reply_count": 0,507          "reply_to_post_number": null,508          "quote_count": 0,509          "incoming_link_count": 22,510          "reads": 6,511          "readers_count": 5,512          "score": 96.2,513          "yours": false,514          "topic_id": 220725,515          "topic_slug": "memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate",516          "display_username": "Jie Huang",517          "primary_group_name": null,518          "flair_name": null,519          "flair_url": null,520          "flair_bg_color": null,521          "flair_color": null,522          "flair_group_id": null,523          "badges_granted": [],524          "version": 2,525          "can_edit": false,526          "can_delete": false,527          "can_recover": false,528          "can_see_hidden_post": false,529          "can_wiki": false,530          "read": true,531          "user_title": "",532          "bookmarked": false,533          "actions_summary": [],534          "moderator": false,535          "admin": false,536          "staff": false,537          "user_id": 83877,538          "hidden": false,539          "trust_level": 1,540          "deleted_at": null,541          "user_deleted": false,542          "edit_reason": null,543          "can_view_edit_history": true,544          "wiki": false,545          "post_url": "/t/memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate/220725/1",546          "can_accept_answer": false,547          "can_unaccept_answer": false,548          "accepted_answer": false,549          "topic_accepted_answer": null,550          "can_vote": false551        },552        {553          "id": 471663,554          "name": "",555          "username": "ptrblck",556          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",557          "created_at": "2025-06-11T14:22:35.617Z",558          "cooked": "<p>It seems you are using different libs in your envs (PyTorch, accelerate). Do you see the same behavior on your 4090 when updating to the latest stack?</p>",559          "post_number": 2,560          "post_type": 1,561          "posts_count": 3,562          "updated_at": "2025-06-11T14:22:35.617Z",563          "reply_count": 1,564          "reply_to_post_number": null,565          "quote_count": 0,566          "incoming_link_count": 0,567          "reads": 4,568          "readers_count": 3,569          "score": 5.8,570          "yours": false,571          "topic_id": 220725,572          "topic_slug": "memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate",573          "display_username": "",574          "primary_group_name": null,575          "flair_name": null,576          "flair_url": null,577          "flair_bg_color": null,578          "flair_color": null,579          "flair_group_id": null,580          "badges_granted": [],581          "version": 1,582          "can_edit": false,583          "can_delete": false,584          "can_recover": false,585          "can_see_hidden_post": false,586          "can_wiki": false,587          "read": true,588          "user_title": "",589          "bookmarked": false,590          "actions_summary": [],591          "moderator": true,592          "admin": true,593          "staff": true,594          "user_id": 3534,595          "hidden": false,596          "trust_level": 2,597          "deleted_at": null,598          "user_deleted": false,599          "edit_reason": null,600          "can_view_edit_history": true,601          "wiki": false,602          "post_url": "/t/memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate/220725/2",603          "can_accept_answer": false,604          "can_unaccept_answer": false,605          "accepted_answer": false,606          "topic_accepted_answer": null607        },608        {609          "id": 471665,610          "name": "Jie Huang",611          "username": "shangxiaaabb",612          "avatar_template": "/user_avatar/discuss.pytorch.org/shangxiaaabb/{size}/76694_2.png",613          "created_at": "2025-06-11T14:26:13.424Z",614          "cooked": "<p>When I run the code on the <strong>4090</strong>, the GPU memory usage remains stable at around <strong>29GB</strong> throughout training, unlike on the <strong>A100</strong> where it keeps increasing over epochs.<br>\n<div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/3X/1/f/1f887098b4e2a32f81998e4c5b42736d5652fef7.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/1f887098b4e2a32f81998e4c5b42736d5652fef7\" title=\"image\"><img src=\"https://discuss.pytorch.org/uploads/default/original/3X/1/f/1f887098b4e2a32f81998e4c5b42736d5652fef7.png\" alt=\"image\" data-base62-sha1=\"4uX7BahLGrTb7FGCOlRfceDkOnd\" width=\"690\" height=\"383\" data-dominant-color=\"EEE9E9\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">image</span><span class=\"informations\">770×428 12.4 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div></p>\n<p><strong>Wait I try it</strong></p>",615          "post_number": 3,616          "post_type": 1,617          "posts_count": 3,618          "updated_at": "2025-06-11T14:27:35.379Z",619          "reply_count": 0,620          "reply_to_post_number": 2,621          "quote_count": 0,622          "incoming_link_count": 3,623          "reads": 4,624          "readers_count": 3,625          "score": 10.8,626          "yours": false,627          "topic_id": 220725,628          "topic_slug": "memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate",629          "display_username": "Jie Huang",630          "primary_group_name": null,631          "flair_name": null,632          "flair_url": null,633          "flair_bg_color": null,634          "flair_color": null,635          "flair_group_id": null,636          "badges_granted": [],637          "version": 3,638          "can_edit": false,639          "can_delete": false,640          "can_recover": false,641          "can_see_hidden_post": false,642          "can_wiki": false,643          "read": true,644          "user_title": "",645          "reply_to_user": {646            "id": 3534,647            "username": "ptrblck",648            "name": "",649            "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"650          },651          "bookmarked": false,652          "actions_summary": [],653          "moderator": false,654          "admin": false,655          "staff": false,656          "user_id": 83877,657          "hidden": false,658          "trust_level": 1,659          "deleted_at": null,660          "user_deleted": false,661          "edit_reason": null,662          "can_view_edit_history": true,663          "wiki": false,664          "post_url": "/t/memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate/220725/3",665          "can_accept_answer": false,666          "can_unaccept_answer": false,667          "accepted_answer": false,668          "topic_accepted_answer": null669        }670      ],671      "stream": [672        471661,673        471663,674        471665675      ]676    },677    "timeline_lookup": [678      [679        1,680        136681      ]682    ],683    "suggested_topics": [684      {685        "fancy_title": "Torch::jit::load error file_name!=nullptr",686        "id": 217824,687        "title": "Torch::jit::load error file_name!=nullptr",688        "slug": "torch-load-error-file-name-nullptr",689        "posts_count": 1,690        "reply_count": 0,691        "highest_post_number": 1,692        "image_url": null,693        "created_at": "2025-03-13T22:18:25.486Z",694        "last_posted_at": "2025-03-13T22:18:25.523Z",695        "bumped": true,696        "bumped_at": "2025-03-13T22:18:25.523Z",697        "archetype": "regular",698        "unseen": false,699        "pinned": false,700        "unpinned": null,701        "visible": true,702        "closed": false,703        "archived": false,704        "bookmarked": null,705        "liked": null,706        "unicode_title": "Torch::jit::load error file_name!=nullptr",707        "tags_descriptions": {},708        "like_count": 0,709        "views": 37,710        "category_id": 1,711        "featured_link": null,712        "has_accepted_answer": false,713        "posters": [714          {715            "extras": "latest single",716            "description": "Original Poster, Most Recent Poster",717            "user": {718              "id": 83260,719              "username": "Sanjib",720              "name": "Sanjib",721              "avatar_template": "/letter_avatar_proxy/v4/letter/s/b4bc9f/{size}.png",722              "trust_level": 1723            }724          }725        ]726      },727      {728        "fancy_title": "Segmentation fault in 50K x 50K torch.matmul (fixed)",729        "id": 218513,730        "title": "Segmentation fault in 50K x 50K torch.matmul (fixed)",731        "slug": "segmentation-fault-in-50k-x-50k-torch-matmul-fixed",732        "posts_count": 2,733        "reply_count": 0,734        "highest_post_number": 2,735        "image_url": null,736        "created_at": "2025-04-02T05:10:32.606Z",737        "last_posted_at": "2025-04-02T14:44:43.440Z",738        "bumped": true,739        "bumped_at": "2025-04-02T14:44:43.440Z",740        "archetype": "regular",741        "unseen": false,742        "pinned": false,743        "unpinned": null,744        "visible": true,745        "closed": false,746        "archived": false,747        "bookmarked": null,748        "liked": null,749        "tags_descriptions": {},750        "like_count": 0,751        "views": 40,752        "category_id": 1,753        "featured_link": null,754        "has_accepted_answer": false,755        "posters": [756          {757            "extras": "latest single",758            "description": "Original Poster, Most Recent Poster",759            "user": {760              "id": 77475,761              "username": "Sanjeev_Rampal",762              "name": "Sanjeev Rampal",763              "avatar_template": "/user_avatar/discuss.pytorch.org/sanjeev_rampal/{size}/62377_2.png",764              "trust_level": 1765            }766          }767        ]768      },769      {770        "fancy_title": "How can I save the proper embeddings and weights after training?",771        "id": 212912,772        "title": "How can I save the proper embeddings and weights after training?",773        "slug": "how-can-i-save-the-proper-embeddings-and-weights-after-training",774        "posts_count": 2,775        "reply_count": 0,776        "highest_post_number": 2,777        "image_url": null,778        "created_at": "2024-11-13T08:45:07.754Z",779        "last_posted_at": "2024-11-13T12:37:00.417Z",780        "bumped": true,781        "bumped_at": "2024-11-13T15:11:25.864Z",782        "archetype": "regular",783        "unseen": false,784        "pinned": false,785        "unpinned": null,786        "visible": true,787        "closed": false,788        "archived": false,789        "bookmarked": null,790        "liked": null,791        "tags_descriptions": {},792        "like_count": 0,793        "views": 68,794        "category_id": 1,795        "featured_link": null,796        "has_accepted_answer": false,797        "posters": [798          {799            "extras": "latest single",800            "description": "Original Poster, Most Recent Poster",801            "user": {802              "id": 72736,803              "username": "songsong0425",804              "name": "Songyeon Lee",805              "avatar_template": "/user_avatar/discuss.pytorch.org/songsong0425/{size}/67200_2.png",806              "trust_level": 1807            }808          }809        ]810      },811      {812        "fancy_title": "Does PyTorch `.to(device)` propagate gradients back to original device?",813        "id": 217641,814        "title": "Does PyTorch `.to(device)` propagate gradients back to original device?",815        "slug": "does-pytorch-to-device-propagate-gradients-back-to-original-device",816        "posts_count": 5,817        "reply_count": 1,818        "highest_post_number": 5,819        "image_url": null,820        "created_at": "2025-03-10T05:14:24.681Z",821        "last_posted_at": "2025-03-12T07:57:27.711Z",822        "bumped": true,823        "bumped_at": "2025-03-12T07:57:27.711Z",824        "archetype": "regular",825        "unseen": false,826        "pinned": false,827        "unpinned": null,828        "visible": true,829        "closed": false,830        "archived": false,831        "bookmarked": null,832        "liked": null,833        "tags_descriptions": {},834        "like_count": 0,835        "views": 195,836        "category_id": 1,837        "featured_link": null,838        "has_accepted_answer": false,839        "posters": [840          {841            "extras": "latest",842            "description": "Original Poster, Most Recent Poster",843            "user": {844              "id": 83164,845              "username": "gammag4",846              "name": "Gabriel Maia",847              "avatar_template": "/user_avatar/discuss.pytorch.org/gammag4/{size}/76063_2.png",848              "trust_level": 1849            }850          },851          {852            "extras": null,853            "description": "Frequent Poster",854            "user": {855              "id": 64488,856              "username": "Naming-isDifficult",857              "name": "",858              "avatar_template": "/user_avatar/discuss.pytorch.org/naming-isdifficult/{size}/58644_2.png",859              "trust_level": 2860            }861          },862          {863            "extras": null,864            "description": "Frequent Poster",865            "user": {866              "id": 3534,867              "username": "ptrblck",868              "name": "",869              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",870              "admin": true,871              "moderator": true,872              "trust_level": 2873            }874          }875        ]876      },877      {878        "fancy_title": "Why is torch.rand different on different devices for big indices",879        "id": 219976,880        "title": "Why is torch.rand different on different devices for big indices",881        "slug": "why-is-torch-rand-different-on-different-devices-for-big-indices",882        "posts_count": 2,883        "reply_count": 0,884        "highest_post_number": 2,885        "image_url": null,886        "created_at": "2025-05-13T03:29:50.431Z",887        "last_posted_at": "2025-05-13T10:56:45.681Z",888        "bumped": true,889        "bumped_at": "2025-05-13T10:56:45.681Z",890        "archetype": "regular",891        "unseen": false,892        "pinned": false,893        "unpinned": null,894        "visible": true,895        "closed": false,896        "archived": false,897        "bookmarked": null,898        "liked": null,899        "tags_descriptions": {},900        "like_count": 1,901        "views": 39,902        "category_id": 1,903        "featured_link": null,904        "has_accepted_answer": false,905        "posters": [906          {907            "extras": null,908            "description": "Original Poster",909            "user": {910              "id": 53070,911              "username": "Jackmin801",912              "name": "Jackmin801",913              "avatar_template": "/user_avatar/discuss.pytorch.org/jackmin801/{size}/46475_2.png",914              "trust_level": 1915            }916          },917          {918            "extras": "latest",919            "description": "Most Recent Poster",920            "user": {921              "id": 3534,922              "username": "ptrblck",923              "name": "",924              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",925              "admin": true,926              "moderator": true,927              "trust_level": 2928            }929          }930        ]931      }932    ],933    "tags_descriptions": {},934    "fancy_title": "Memory issues when running the same code on A100 and 4090Ti when using accelerate",935    "id": 220725,936    "title": "Memory issues when running the same code on A100 and 4090Ti when using accelerate",937    "posts_count": 3,938    "created_at": "2025-06-11T14:11:37.200Z",939    "views": 41,940    "reply_count": 1,941    "like_count": 0,942    "last_posted_at": "2025-06-11T14:26:13.424Z",943    "visible": true,944    "closed": false,945    "archived": false,946    "has_summary": false,947    "archetype": "regular",948    "slug": "memory-issues-when-running-the-same-code-on-a100-and-4090ti-when-using-accelerate",949    "category_id": 1,950    "word_count": 1210,951    "deleted_at": null,952    "user_id": 83877,953    "featured_link": null,954    "pinned_globally": false,955    "pinned_at": null,956    "pinned_until": null,957    "image_url": null,958    "slow_mode_seconds": 0,959    "draft": null,960    "draft_key": "topic_220725",961    "draft_sequence": null,962    "unpinned": null,963    "pinned": false,964    "current_post_number": 1,965    "highest_post_number": 3,966    "deleted_by": null,967    "actions_summary": [968      {969        "id": 4,970        "count": 0,971        "hidden": false,972        "can_act": false973      },974      {975        "id": 8,976        "count": 0,977        "hidden": false,978        "can_act": false979      },980      {981        "id": 10,982        "count": 0,983        "hidden": false,984        "can_act": false985      },986      {987        "id": 7,988        "count": 0,989        "hidden": false,990        "can_act": false991      }992    ],993    "chunk_size": 20,994    "bookmarked": false,995    "topic_timer": null,996    "message_bus_last_id": 0,997    "participant_count": 2,998    "show_read_indicator": false,999    "thumbnails": null,1000    "slow_mode_enabled_until": null,1001    "can_vote": false,1002    "vote_count": 0,1003    "user_voted": false,1004    "discourse_zendesk_plugin_zendesk_id": null,1005    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",1006    "details": {1007      "can_edit": false,1008      "notification_level": 1,1009      "participants": [1010        {1011          "id": 83877,1012          "username": "shangxiaaabb",1013          "name": "Jie Huang",1014          "avatar_template": "/user_avatar/discuss.pytorch.org/shangxiaaabb/{size}/76694_2.png",1015          "post_count": 2,1016          "primary_group_name": null,1017          "flair_name": null,1018          "flair_url": null,1019          "flair_color": null,1020          "flair_bg_color": null,1021          "flair_group_id": null,1022          "trust_level": 11023        },1024        {1025          "id": 3534,1026          "username": "ptrblck",1027          "name": "",1028          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1029          "post_count": 1,1030          "primary_group_name": null,1031          "flair_name": null,1032          "flair_url": null,1033          "flair_color": null,1034          "flair_bg_color": null,1035          "flair_group_id": null,1036          "admin": true,1037          "moderator": true,1038          "trust_level": 21039        }1040      ],1041      "created_by": {1042        "id": 83877,1043        "username": "shangxiaaabb",1044        "name": "Jie Huang",1045        "avatar_template": "/user_avatar/discuss.pytorch.org/shangxiaaabb/{size}/76694_2.png"1046      },1047      "last_poster": {1048        "id": 83877,1049        "username": "shangxiaaabb",1050        "name": "Jie Huang",1051        "avatar_template": "/user_avatar/discuss.pytorch.org/shangxiaaabb/{size}/76694_2.png"1052      }1053    },1054    "bookmarks": []1055  },1056  {1057    "post_stream": {1058      "posts": [1059        {1060          "id": 468976,1061          "name": "nafe alhmdan",1062          "username": "nafe_alhmdan",1063          "avatar_template": "/user_avatar/discuss.pytorch.org/nafe_alhmdan/{size}/75984_2.png",1064          "created_at": "2025-04-10T16:04:34.876Z",1065          "cooked": "<p>C:\\Users\\USER\\Desktop\\RVC1006Nvidia&gt;runtime\\python.exe infer-web.py --pycmd runtime\\python.exe --port 7897<br>\nC:\\Users\\USER\\Desktop\\RVC1006Nvidia\\runtime\\lib\\site-packages\\torch\\cuda_<em>init</em>_.py:173: UserWarning:<br>\nNVIDIA GeForce RTX 5090 with CUDA capability sm_120 is not compatible with the current PyTorch installation.<br>\nThe current PyTorch install supports CUDA capabilities sm_37 sm_50 sm_60 sm_61 sm_70 sm_75 sm_80 sm_86 sm_90 compute_37.<br>\nIf you want to use the NVIDIA GeForce RTX 5090 GPU with PyTorch, please check the instructions at <a href=\"https://pytorch.org/get-started/locally/\" class=\"inline-onebox\" rel=\"noopener nofollow ugc\">Start Locally | PyTorch</a></p>\n<p>warnings.warn(incompatible_device_warn.format(device_name, capability, \" \".join(arch_list), device_name))<br>\n2025-04-10 18:55:22 | INFO | configs.config | Found GPU NVIDIA GeForce RTX 5090<br>\nis_half:True, device:cuda:0</p>",1066          "post_number": 1,1067          "post_type": 1,1068          "posts_count": 11,1069          "updated_at": "2025-04-10T16:04:34.876Z",1070          "reply_count": 0,1071          "reply_to_post_number": null,1072          "quote_count": 0,1073          "incoming_link_count": 10949,1074          "reads": 36,1075          "readers_count": 35,1076          "score": 53597.2,1077          "yours": false,1078          "topic_id": 218954,1079          "topic_slug": "nvidia-geforce-rtx-5090",1080          "display_username": "nafe alhmdan",1081          "primary_group_name": null,1082          "flair_name": null,1083          "flair_url": null,1084          "flair_bg_color": null,1085          "flair_color": null,1086          "flair_group_id": null,1087          "badges_granted": [],1088          "version": 1,1089          "can_edit": false,1090          "can_delete": false,1091          "can_recover": false,1092          "can_see_hidden_post": false,1093          "can_wiki": false,1094          "link_counts": [1095            {1096              "url": "https://pytorch.org/get-started/locally/",1097              "internal": false,1098              "reflection": false,1099              "title": "Start Locally | PyTorch",1100              "clicks": 2411101            }1102          ],1103          "read": true,1104          "user_title": null,1105          "bookmarked": false,1106          "actions_summary": [],1107          "moderator": false,1108          "admin": false,1109          "staff": false,1110          "user_id": 83067,1111          "hidden": false,1112          "trust_level": 1,1113          "deleted_at": null,1114          "user_deleted": false,1115          "edit_reason": null,1116          "can_view_edit_history": true,1117          "wiki": false,1118          "post_url": "/t/nvidia-geforce-rtx-5090/218954/1",1119          "can_accept_answer": false,1120          "can_unaccept_answer": false,1121          "accepted_answer": false,1122          "topic_accepted_answer": null,1123          "can_vote": false1124        },1125        {1126          "id": 468977,1127          "name": "",1128          "username": "ptrblck",1129          "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1130          "created_at": "2025-04-10T16:06:14.959Z",1131          "cooked": "<p>Install the latest nightly binaries with CUDA 12.8 and it will work as the Blackwell support requires CUDA &gt;= 12.8.</p>",1132          "post_number": 2,1133          "post_type": 1,1134          "posts_count": 11,1135          "updated_at": "2025-04-10T16:06:14.959Z",1136          "reply_count": 0,1137          "reply_to_post_number": null,1138          "quote_count": 0,1139          "incoming_link_count": 36,1140          "reads": 38,1141          "readers_count": 37,1142          "score": 187.6,1143          "yours": false,1144          "topic_id": 218954,1145          "topic_slug": "nvidia-geforce-rtx-5090",1146          "display_username": "",1147          "primary_group_name": null,1148          "flair_name": null,1149          "flair_url": null,1150          "flair_bg_color": null,1151          "flair_color": null,1152          "flair_group_id": null,1153          "badges_granted": [],1154          "version": 1,1155          "can_edit": false,1156          "can_delete": false,1157          "can_recover": false,1158          "can_see_hidden_post": false,1159          "can_wiki": false,1160          "read": true,1161          "user_title": "",1162          "bookmarked": false,1163          "actions_summary": [],1164          "moderator": true,1165          "admin": true,1166          "staff": true,1167          "user_id": 3534,1168          "hidden": false,1169          "trust_level": 2,1170          "deleted_at": null,1171          "user_deleted": false,1172          "edit_reason": null,1173          "can_view_edit_history": true,1174          "wiki": false,1175          "post_url": "/t/nvidia-geforce-rtx-5090/218954/2",1176          "can_accept_answer": false,1177          "can_unaccept_answer": false,1178          "accepted_answer": false,1179          "topic_accepted_answer": null1180        },1181        {1182          "id": 468979,1183          "name": "nafe alhmdan",1184          "username": "nafe_alhmdan",1185          "avatar_template": "/user_avatar/discuss.pytorch.org/nafe_alhmdan/{size}/75984_2.png",1186          "created_at": "2025-04-10T16:42:59.136Z",1187          "cooked": "<p>Python 3.12.8 (tags/v3.12.8:2dc476b, Dec  3 2024, 19:30:04) [MSC v.1942 64 bit (AMD64)] on win32<br>\nType “help”, “copyright”, “credits” or “license” for more information.</p>\n<blockquote>\n<blockquote>\n<blockquote>\n<p>import torch<br>\nprint(torch.<strong>version</strong>)<br>\n2.8.0.dev20250408+cu128</p>\n</blockquote>\n</blockquote>\n</blockquote>",1188          "post_number": 3,1189          "post_type": 1,1190          "posts_count": 11,1191          "updated_at": "2025-04-10T16:42:59.136Z",1192          "reply_count": 0,1193          "reply_to_post_number": null,1194          "quote_count": 0,1195          "incoming_link_count": 46,1196          "reads": 36,1197          "readers_count": 35,1198          "score": 237.2,1199          "yours": false,1200          "topic_id": 218954,

Showing the first 1,200 of 65950 lines. Download the file for the rest.