Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 133885,7 "name": "saman",8 "username": "samanmo",9 "avatar_template": "/user_avatar/discuss.pytorch.org/samanmo/{size}/45907_2.png",10 "created_at": "2019-09-09T09:52:18.282Z",11 "cooked": "<p>I am using SubsetRandomSampler() class in pytorch to train on a subset of my data set,<br>\nI have got these problems:</p>\n<ol>\n<li>as dataset I use CocoDetection and my data loader seems not returning labels for all objects.</li>\n<li>how could I get the indices randomly chosen by SubsetRandomSampler from the list of indices that I provided at each batch run?</li>\n</ol>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 1,15 "updated_at": "2019-09-09T09:53:09.135Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 7,20 "reads": 8,21 "readers_count": 7,22 "score": 36.6,23 "yours": false,24 "topic_id": 55523,25 "topic_slug": "subsetrandomsampler",26 "display_username": "saman",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 3,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "read": true,41 "user_title": null,42 "bookmarked": false,43 "actions_summary": [],44 "moderator": false,45 "admin": false,46 "staff": false,47 "user_id": 22383,48 "hidden": false,49 "trust_level": 1,50 "deleted_at": null,51 "user_deleted": false,52 "edit_reason": null,53 "can_view_edit_history": true,54 "wiki": false,55 "post_url": "/t/subsetrandomsampler/55523/1",56 "can_accept_answer": false,57 "can_unaccept_answer": false,58 "accepted_answer": false,59 "topic_accepted_answer": null,60 "can_vote": false61 }62 ],63 "stream": [64 13388565 ]66 },67 "timeline_lookup": [68 [69 1,70 223971 ]72 ],73 "suggested_topics": [74 {75 "fancy_title": "Loss not Decreasing while training UNET",76 "id": 219167,77 "title": "Loss not Decreasing while training UNET",78 "slug": "loss-not-decreasing-while-training-unet",79 "posts_count": 4,80 "reply_count": 1,81 "highest_post_number": 4,82 "image_url": null,83 "created_at": "2025-04-16T17:26:41.889Z",84 "last_posted_at": "2025-05-26T19:08:48.848Z",85 "bumped": true,86 "bumped_at": "2025-05-26T19:08:48.848Z",87 "archetype": "regular",88 "unseen": false,89 "pinned": false,90 "unpinned": null,91 "visible": true,92 "closed": false,93 "archived": false,94 "bookmarked": null,95 "liked": null,96 "tags_descriptions": {},97 "like_count": 0,98 "views": 143,99 "category_id": 5,100 "featured_link": null,101 "has_accepted_answer": false,102 "posters": [103 {104 "extras": null,105 "description": "Original Poster",106 "user": {107 "id": 83857,108 "username": "Jaskeerat",109 "name": "Jaskeerat",110 "avatar_template": "/user_avatar/discuss.pytorch.org/jaskeerat/{size}/76675_2.png",111 "trust_level": 0112 }113 },114 {115 "extras": null,116 "description": "Frequent Poster",117 "user": {118 "id": 18088,119 "username": "KFrank",120 "name": "K. Frank",121 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",122 "trust_level": 2123 }124 },125 {126 "extras": "latest",127 "description": "Most Recent Poster",128 "user": {129 "id": 75120,130 "username": "ajayrkumar",131 "name": "Ajay Rajendra Kumar",132 "avatar_template": "/user_avatar/discuss.pytorch.org/ajayrkumar/{size}/77139_2.png",133 "trust_level": 1134 }135 }136 ]137 },138 {139 "fancy_title": "Applying transformation to bounding boxes is not producing proper result",140 "id": 216126,141 "title": "Applying transformation to bounding boxes is not producing proper result",142 "slug": "applying-transformation-to-bounding-boxes-is-not-producing-proper-result",143 "posts_count": 3,144 "reply_count": 1,145 "highest_post_number": 3,146 "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/a/1/a1c0ea31ca20f4f023bbcbd40d2f5a8e33975582.jpeg",147 "created_at": "2025-02-01T18:00:17.089Z",148 "last_posted_at": "2025-02-02T05:32:01.128Z",149 "bumped": true,150 "bumped_at": "2025-02-02T05:32:01.128Z",151 "archetype": "regular",152 "unseen": false,153 "pinned": false,154 "unpinned": null,155 "visible": true,156 "closed": false,157 "archived": false,158 "bookmarked": null,159 "liked": null,160 "tags_descriptions": {},161 "like_count": 0,162 "views": 111,163 "category_id": 5,164 "featured_link": null,165 "has_accepted_answer": true,166 "posters": [167 {168 "extras": "latest",169 "description": "Original Poster, Most Recent Poster",170 "user": {171 "id": 82180,172 "username": "Amit_Sur",173 "name": "Amit Sur",174 "avatar_template": "/user_avatar/discuss.pytorch.org/amit_sur/{size}/75195_2.png",175 "trust_level": 1176 }177 },178 {179 "extras": null,180 "description": "Frequent Poster, Accepted Answer",181 "user": {182 "id": 3534,183 "username": "ptrblck",184 "name": "",185 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",186 "admin": true,187 "moderator": true,188 "trust_level": 2189 }190 }191 ]192 },193 {194 "fancy_title": "How to load the best weights of yolov6 after training?",195 "id": 216017,196 "title": "How to load the best weights of yolov6 after training?",197 "slug": "how-to-load-the-best-weights-of-yolov6-after-training",198 "posts_count": 2,199 "reply_count": 0,200 "highest_post_number": 2,201 "image_url": null,202 "created_at": "2025-01-29T06:31:26.226Z",203 "last_posted_at": "2025-01-29T22:50:32.204Z",204 "bumped": true,205 "bumped_at": "2025-01-29T22:50:32.204Z",206 "archetype": "regular",207 "unseen": false,208 "pinned": false,209 "unpinned": null,210 "visible": true,211 "closed": false,212 "archived": false,213 "bookmarked": null,214 "liked": null,215 "tags_descriptions": {},216 "like_count": 0,217 "views": 116,218 "category_id": 5,219 "featured_link": null,220 "has_accepted_answer": false,221 "posters": [222 {223 "extras": null,224 "description": "Original Poster",225 "user": {226 "id": 78758,227 "username": "edgy_sharma",228 "name": "kanchi sharma",229 "avatar_template": "/user_avatar/discuss.pytorch.org/edgy_sharma/{size}/72607_2.png",230 "trust_level": 1231 }232 },233 {234 "extras": "latest",235 "description": "Most Recent Poster",236 "user": {237 "id": 3534,238 "username": "ptrblck",239 "name": "",240 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",241 "admin": true,242 "moderator": true,243 "trust_level": 2244 }245 }246 ]247 },248 {249 "fancy_title": "Optical Flow through backward warping?",250 "id": 217655,251 "title": "Optical Flow through backward warping?",252 "slug": "optical-flow-through-backward-warping",253 "posts_count": 1,254 "reply_count": 0,255 "highest_post_number": 1,256 "image_url": null,257 "created_at": "2025-03-10T09:23:24.572Z",258 "last_posted_at": "2025-03-10T09:23:24.610Z",259 "bumped": true,260 "bumped_at": "2025-03-10T09:23:24.610Z",261 "archetype": "regular",262 "unseen": false,263 "pinned": false,264 "unpinned": null,265 "visible": true,266 "closed": false,267 "archived": false,268 "bookmarked": null,269 "liked": null,270 "tags_descriptions": {},271 "like_count": 0,272 "views": 58,273 "category_id": 5,274 "featured_link": null,275 "has_accepted_answer": false,276 "posters": [277 {278 "extras": "latest single",279 "description": "Original Poster, Most Recent Poster",280 "user": {281 "id": 83167,282 "username": "ppoyazero",283 "name": "",284 "avatar_template": "/user_avatar/discuss.pytorch.org/ppoyazero/{size}/76068_2.png",285 "trust_level": 1286 }287 }288 ]289 },290 {291 "fancy_title": "Both training and validation loss not reducing",292 "id": 219947,293 "title": "Both training and validation loss not reducing",294 "slug": "both-training-and-validation-loss-not-reducing",295 "posts_count": 1,296 "reply_count": 0,297 "highest_post_number": 1,298 "image_url": null,299 "created_at": "2025-05-12T06:58:44.939Z",300 "last_posted_at": "2025-05-12T06:58:44.989Z",301 "bumped": true,302 "bumped_at": "2025-05-12T06:58:44.989Z",303 "archetype": "regular",304 "unseen": false,305 "pinned": false,306 "unpinned": null,307 "visible": true,308 "closed": false,309 "archived": false,310 "bookmarked": null,311 "liked": null,312 "tags_descriptions": {},313 "like_count": 0,314 "views": 40,315 "category_id": 5,316 "featured_link": null,317 "has_accepted_answer": false,318 "posters": [319 {320 "extras": "latest single",321 "description": "Original Poster, Most Recent Poster",322 "user": {323 "id": 84239,324 "username": "Soham_Bhaumik",325 "name": "Soham Bhaumik",326 "avatar_template": "/user_avatar/discuss.pytorch.org/soham_bhaumik/{size}/75355_2.png",327 "trust_level": 1328 }329 }330 ]331 }332 ],333 "tags_descriptions": {},334 "fancy_title": "SubsetRandomSampler",335 "id": 55523,336 "title": "SubsetRandomSampler",337 "posts_count": 1,338 "created_at": "2019-09-09T09:52:18.207Z",339 "views": 260,340 "reply_count": 0,341 "like_count": 0,342 "last_posted_at": "2019-09-09T09:52:18.282Z",343 "visible": true,344 "closed": false,345 "archived": false,346 "has_summary": false,347 "archetype": "regular",348 "slug": "subsetrandomsampler",349 "category_id": 5,350 "word_count": 61,351 "deleted_at": null,352 "user_id": 22383,353 "featured_link": null,354 "pinned_globally": false,355 "pinned_at": null,356 "pinned_until": null,357 "image_url": null,358 "slow_mode_seconds": 0,359 "draft": null,360 "draft_key": "topic_55523",361 "draft_sequence": null,362 "unpinned": null,363 "pinned": false,364 "current_post_number": 1,365 "highest_post_number": 1,366 "deleted_by": null,367 "actions_summary": [368 {369 "id": 4,370 "count": 0,371 "hidden": false,372 "can_act": false373 },374 {375 "id": 8,376 "count": 0,377 "hidden": false,378 "can_act": false379 },380 {381 "id": 10,382 "count": 0,383 "hidden": false,384 "can_act": false385 },386 {387 "id": 7,388 "count": 0,389 "hidden": false,390 "can_act": false391 }392 ],393 "chunk_size": 20,394 "bookmarked": false,395 "topic_timer": null,396 "message_bus_last_id": 0,397 "participant_count": 1,398 "show_read_indicator": false,399 "thumbnails": null,400 "slow_mode_enabled_until": null,401 "can_vote": false,402 "vote_count": 0,403 "user_voted": false,404 "discourse_zendesk_plugin_zendesk_id": null,405 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",406 "details": {407 "can_edit": false,408 "notification_level": 1,409 "participants": [410 {411 "id": 22383,412 "username": "samanmo",413 "name": "saman",414 "avatar_template": "/user_avatar/discuss.pytorch.org/samanmo/{size}/45907_2.png",415 "post_count": 1,416 "primary_group_name": null,417 "flair_name": null,418 "flair_url": null,419 "flair_color": null,420 "flair_bg_color": null,421 "flair_group_id": null,422 "trust_level": 1423 }424 ],425 "created_by": {426 "id": 22383,427 "username": "samanmo",428 "name": "saman",429 "avatar_template": "/user_avatar/discuss.pytorch.org/samanmo/{size}/45907_2.png"430 },431 "last_poster": {432 "id": 22383,433 "username": "samanmo",434 "name": "saman",435 "avatar_template": "/user_avatar/discuss.pytorch.org/samanmo/{size}/45907_2.png"436 }437 },438 "bookmarks": []439 },440 {441 "post_stream": {442 "posts": [443 {444 "id": 133878,445 "name": "",446 "username": "davidlee",447 "avatar_template": "/user_avatar/discuss.pytorch.org/davidlee/{size}/12812_2.png",448 "created_at": "2019-09-09T08:27:38.250Z",449 "cooked": "<p>Hello PyTorch!<br>\nI’m training Seq2Seq model with 2080 Ti and I cannot use GPU fully right now.</p>\n<ul>\n<li>Input dimension, output dimension = 1</li>\n<li>Hidden dimension = 20</li>\n<li>Sequence length (in and out) = 10</li>\n<li>batch size = 100, totally 200 batches<br>\nHere is my code for <strong>Dataset, Model and Training</strong>\n</li>\n</ul>\n<p><strong>I’m using single GPU and memory-usage and utilization are 951Mb and 22% respectivly.</strong></p>\n<p>Please give any advices or tips for using Memory and GPU fully!!!</p>\n<p>Thanks.</p>\n<hr>\n<p>dataset = myDataset()<br>\ntrain_loader = DataLoader(dataset=dataset, batch_size=100, shuffle=True, drop_last=True, num_workers=4)</p>\n<hr>\n<p>class Encoder(nn.Module):</p>\n<pre><code>def __init__(self, input_dim, hidden_dim, num_layers=1, dropout=0, bidirectional=False):\n super(Encoder, self).__init__()\n self.encoder = nn.LSTM(input_dim, hidden_dim, num_layers=num_layers, dropout=dropout, bidirectional=bidirectional)\n \ndef forward(self, x, hidden):\n encoder_output, encoder_state = self.encoder(x, hidden)\n return encoder_output, encoder_state\n</code></pre>\n<hr>\n<p>class Decoder(nn.Module):</p>\n<pre><code>def __init__(self, input_dim, hidden_dim, output_dim, num_layers=1, dropout=0, bidirectional=False):\n super(Decoder, self).__init__()\n self.decoder = nn.LSTM(input_dim, hidden_dim, num_layers=num_layers, dropout=dropout, bidirectional=bidirectional)\n self.linear = nn.Linear(hidden_dim, output_dim)\n\ndef forward(self, x, hidden):\n decoder_output, next_hidden = self.decoder(x, hidden)\n \n outputs = []\n for i in range(decoder_output.size()[1]):\n outputs += [self.linear(decoder_output[:, i, :])]\n return torch.stack(outputs, dim=1).squeeze(), decoder_output, next_hidden\n</code></pre>\n<hr>\n<p>class Model(nn.Module):</p>\n<pre><code>def __init__(self, input_dim, hidden_dim, output_dim, num_layers=1, output_length=10):\n super(Model, self).__init__()\n self.encoder = Encoder(input_dim, hidden_dim, num_layers=num_layers)\n self.decoder = Decoder(hidden_dim, hidden_dim, output_dim, num_layers=num_layers)\n self.output_length = output_length\n self.num_layers = num_layers\n self.hidden_dim = hidden_dim\n \ndef forward(self, x):\n encoder_output, encdoer_state = self.encoder(x, None) \n \n decoder_input = torch.unsqueeze(encoder_output[-1], 0)\n \n seq = []\n next_hidden=None \n next_input = decoder_input\n \n for _ in range(self.output_length):\n output, next_input, next_hidden = self.decoder(next_input, next_hidden)\n seq += [output]\n return torch.stack(seq, dim=0).squeeze(), torch.unsqueeze(encoder_output[-1],0)\n</code></pre>\n<hr>\n<p>loss_func = nn.MSELoss().to(device)<br>\noptimizer = optim.Adam(model.parameters(), lr=0.001)</p>\n<p>epochs = 20</p>\n<p>total_batch = len(train_loader)<br>\nprint(‘total_batch = {}’.format(total_batch))</p>\n<p>model.train()<br>\ntrain_loss = []</p>\n<p>for epoch in range(epochs):</p>\n<pre><code>avg_cost = 0.0\n\nfor nums, data in enumerate(train_loader):\n temp_x, temp_y = data\n \n x = torch.FloatTensor(temp_x)\n y = torch.FloatTensor(temp_y)\n \n x = np.transpose(x, (1,0,2))\n y = np.transpose(y, (1,0,2))\n\n \n optimizer.zero_grad()\n \n prediction, fixed_vector = model(x.to(device)) # rnn output\n prediction = prediction.unsqueeze(2) \n loss = loss_func(prediction, y.to(device))\n \n loss.backward()\n optimizer.step()</code></pre>",450 "post_number": 1,451 "post_type": 1,452 "posts_count": 1,453 "updated_at": "2019-09-09T08:27:38.250Z",454 "reply_count": 0,455 "reply_to_post_number": null,456 "quote_count": 0,457 "incoming_link_count": 53,458 "reads": 7,459 "readers_count": 6,460 "score": 266.4,461 "yours": false,462 "topic_id": 55520,463 "topic_slug": "need-help-about-low-gpu-utilization-when-training-seq2seq",464 "display_username": "",465 "primary_group_name": null,466 "flair_name": null,467 "flair_url": null,468 "flair_bg_color": null,469 "flair_color": null,470 "flair_group_id": null,471 "badges_granted": [],472 "version": 1,473 "can_edit": false,474 "can_delete": false,475 "can_recover": false,476 "can_see_hidden_post": false,477 "can_wiki": false,478 "read": true,479 "user_title": null,480 "bookmarked": false,481 "actions_summary": [],482 "moderator": false,483 "admin": false,484 "staff": false,485 "user_id": 20772,486 "hidden": false,487 "trust_level": 1,488 "deleted_at": null,489 "user_deleted": false,490 "edit_reason": null,491 "can_view_edit_history": true,492 "wiki": false,493 "post_url": "/t/need-help-about-low-gpu-utilization-when-training-seq2seq/55520/1",494 "can_accept_answer": false,495 "can_unaccept_answer": false,496 "accepted_answer": false,497 "topic_accepted_answer": null,498 "can_vote": false499 }500 ],501 "stream": [502 133878503 ]504 },505 "timeline_lookup": [506 [507 1,508 2239509 ]510 ],511 "suggested_topics": [512 {513 "fancy_title": "How to Compare Custom CUDA Gradients with PyTorch’s Autograd Gradients?",514 "id": 213431,515 "title": "How to Compare Custom CUDA Gradients with PyTorch's Autograd Gradients?",516 "slug": "how-to-compare-custom-cuda-gradients-with-pytorchs-autograd-gradients",517 "posts_count": 3,518 "reply_count": 0,519 "highest_post_number": 3,520 "image_url": null,521 "created_at": "2024-11-26T01:04:27.643Z",522 "last_posted_at": "2024-12-13T18:38:05.506Z",523 "bumped": true,524 "bumped_at": "2024-12-13T18:38:05.506Z",525 "archetype": "regular",526 "unseen": false,527 "pinned": false,528 "unpinned": null,529 "visible": true,530 "closed": false,531 "archived": false,532 "bookmarked": null,533 "liked": null,534 "tags_descriptions": {},535 "like_count": 0,536 "views": 143,537 "category_id": 7,538 "featured_link": null,539 "has_accepted_answer": true,540 "posters": [541 {542 "extras": null,543 "description": "Original Poster",544 "user": {545 "id": 81123,546 "username": "OmkarV23",547 "name": "Omkar Vengurlekar",548 "avatar_template": "/user_avatar/discuss.pytorch.org/omkarv23/{size}/74199_2.png",549 "trust_level": 1550 }551 },552 {553 "extras": "latest",554 "description": "Most Recent Poster, Accepted Answer",555 "user": {556 "id": 41396,557 "username": "soulitzer",558 "name": "",559 "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",560 "trust_level": 2561 }562 }563 ]564 },565 {566 "fancy_title": "Monitor optimizer step - Adam",567 "id": 219589,568 "title": "Monitor optimizer step - Adam",569 "slug": "monitor-optimizer-step-adam",570 "posts_count": 2,571 "reply_count": 0,572 "highest_post_number": 3,573 "image_url": null,574 "created_at": "2025-04-29T10:34:48.814Z",575 "last_posted_at": "2025-04-30T10:10:17.457Z",576 "bumped": true,577 "bumped_at": "2025-04-30T10:10:17.457Z",578 "archetype": "regular",579 "unseen": false,580 "pinned": false,581 "unpinned": null,582 "visible": true,583 "closed": false,584 "archived": false,585 "bookmarked": null,586 "liked": null,587 "tags_descriptions": {},588 "like_count": 0,589 "views": 93,590 "category_id": 7,591 "featured_link": null,592 "has_accepted_answer": false,593 "posters": [594 {595 "extras": null,596 "description": "Original Poster",597 "user": {598 "id": 82217,599 "username": "Johannes_Vogt",600 "name": "Johannes Vogt",601 "avatar_template": "/user_avatar/discuss.pytorch.org/johannes_vogt/{size}/75220_2.png",602 "trust_level": 1603 }604 },605 {606 "extras": "latest",607 "description": "Most Recent Poster",608 "user": {609 "id": 75871,610 "username": "qq-me",611 "name": "Ivan Nikishev",612 "avatar_template": "/user_avatar/discuss.pytorch.org/qq-me/{size}/70055_2.png",613 "trust_level": 2614 }615 }616 ]617 },618 {619 "fancy_title": "How to preserve computational graph while initializing a network with weights",620 "id": 217388,621 "title": "How to preserve computational graph while initializing a network with weights",622 "slug": "how-to-preserve-computational-graph-while-initializing-a-network-with-weights",623 "posts_count": 3,624 "reply_count": 1,625 "highest_post_number": 3,626 "image_url": null,627 "created_at": "2025-03-03T16:08:29.829Z",628 "last_posted_at": "2025-03-04T05:38:35.265Z",629 "bumped": true,630 "bumped_at": "2025-03-04T05:38:35.265Z",631 "archetype": "regular",632 "unseen": false,633 "pinned": false,634 "unpinned": null,635 "visible": true,636 "closed": false,637 "archived": false,638 "bookmarked": null,639 "liked": null,640 "tags_descriptions": {},641 "like_count": 0,642 "views": 50,643 "category_id": 7,644 "featured_link": null,645 "has_accepted_answer": true,646 "posters": [647 {648 "extras": "latest",649 "description": "Original Poster, Most Recent Poster",650 "user": {651 "id": 83043,652 "username": "Charley_Xiao",653 "name": "Charley Xiao",654 "avatar_template": "/user_avatar/discuss.pytorch.org/charley_xiao/{size}/75963_2.png",655 "trust_level": 1656 }657 },658 {659 "extras": null,660 "description": "Frequent Poster, Accepted Answer",661 "user": {662 "id": 18088,663 "username": "KFrank",664 "name": "K. Frank",665 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",666 "trust_level": 2667 }668 }669 ]670 },671 {672 "fancy_title": "Computing Gradient of Loss w.r.t Learning Rate",673 "id": 217673,674 "title": "Computing Gradient of Loss w.r.t Learning Rate",675 "slug": "computing-gradient-of-loss-w-r-t-learning-rate",676 "posts_count": 2,677 "reply_count": 0,678 "highest_post_number": 2,679 "image_url": null,680 "created_at": "2025-03-10T15:26:24.072Z",681 "last_posted_at": "2025-03-12T03:24:47.211Z",682 "bumped": true,683 "bumped_at": "2025-03-12T03:24:47.211Z",684 "archetype": "regular",685 "unseen": false,686 "pinned": false,687 "unpinned": null,688 "visible": true,689 "closed": false,690 "archived": false,691 "bookmarked": null,692 "liked": null,693 "tags_descriptions": {},694 "like_count": 0,695 "views": 71,696 "category_id": 7,697 "featured_link": null,698 "has_accepted_answer": false,699 "posters": [700 {701 "extras": null,702 "description": "Original Poster",703 "user": {704 "id": 77481,705 "username": "maticos",706 "name": "Mateo D.",707 "avatar_template": "/user_avatar/discuss.pytorch.org/maticos/{size}/71494_2.png",708 "trust_level": 1709 }710 },711 {712 "extras": "latest",713 "description": "Most Recent Poster",714 "user": {715 "id": 18088,716 "username": "KFrank",717 "name": "K. Frank",718 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",719 "trust_level": 2720 }721 }722 ]723 },724 {725 "fancy_title": "Segfault in autograd after using torch lightning",726 "id": 219723,727 "title": "Segfault in autograd after using torch lightning",728 "slug": "segfault-in-autograd-after-using-torch-lightning",729 "posts_count": 2,730 "reply_count": 0,731 "highest_post_number": 2,732 "image_url": null,733 "created_at": "2025-05-04T04:09:05.621Z",734 "last_posted_at": "2025-05-04T16:37:38.109Z",735 "bumped": true,736 "bumped_at": "2025-05-04T16:37:38.109Z",737 "archetype": "regular",738 "unseen": false,739 "pinned": false,740 "unpinned": null,741 "visible": true,742 "closed": false,743 "archived": false,744 "bookmarked": null,745 "liked": null,746 "tags_descriptions": {},747 "like_count": 0,748 "views": 80,749 "category_id": 7,750 "featured_link": null,751 "has_accepted_answer": false,752 "posters": [753 {754 "extras": null,755 "description": "Original Poster",756 "user": {757 "id": 84135,758 "username": "mxposed",759 "name": "Nick Markov",760 "avatar_template": "/user_avatar/discuss.pytorch.org/mxposed/{size}/76899_2.png",761 "trust_level": 0762 }763 },764 {765 "extras": "latest",766 "description": "Most Recent Poster",767 "user": {768 "id": 3534,769 "username": "ptrblck",770 "name": "",771 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",772 "admin": true,773 "moderator": true,774 "trust_level": 2775 }776 }777 ]778 }779 ],780 "tags_descriptions": {},781 "fancy_title": "Need help about Low GPU Utilization when training Seq2Seq",782 "id": 55520,783 "title": "Need help about Low GPU Utilization when training Seq2Seq",784 "posts_count": 1,785 "created_at": "2019-09-09T08:27:38.189Z",786 "views": 404,787 "reply_count": 0,788 "like_count": 0,789 "last_posted_at": "2019-09-09T08:27:38.250Z",790 "visible": true,791 "closed": false,792 "archived": false,793 "has_summary": false,794 "archetype": "regular",795 "slug": "need-help-about-low-gpu-utilization-when-training-seq2seq",796 "category_id": 7,797 "word_count": 383,798 "deleted_at": null,799 "user_id": 20772,800 "featured_link": null,801 "pinned_globally": false,802 "pinned_at": null,803 "pinned_until": null,804 "image_url": null,805 "slow_mode_seconds": 0,806 "draft": null,807 "draft_key": "topic_55520",808 "draft_sequence": null,809 "unpinned": null,810 "pinned": false,811 "current_post_number": 1,812 "highest_post_number": 1,813 "deleted_by": null,814 "actions_summary": [815 {816 "id": 4,817 "count": 0,818 "hidden": false,819 "can_act": false820 },821 {822 "id": 8,823 "count": 0,824 "hidden": false,825 "can_act": false826 },827 {828 "id": 10,829 "count": 0,830 "hidden": false,831 "can_act": false832 },833 {834 "id": 7,835 "count": 0,836 "hidden": false,837 "can_act": false838 }839 ],840 "chunk_size": 20,841 "bookmarked": false,842 "topic_timer": null,843 "message_bus_last_id": 0,844 "participant_count": 1,845 "show_read_indicator": false,846 "thumbnails": null,847 "slow_mode_enabled_until": null,848 "can_vote": false,849 "vote_count": 0,850 "user_voted": false,851 "discourse_zendesk_plugin_zendesk_id": null,852 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",853 "details": {854 "can_edit": false,855 "notification_level": 1,856 "participants": [857 {858 "id": 20772,859 "username": "davidlee",860 "name": "",861 "avatar_template": "/user_avatar/discuss.pytorch.org/davidlee/{size}/12812_2.png",862 "post_count": 1,863 "primary_group_name": null,864 "flair_name": null,865 "flair_url": null,866 "flair_color": null,867 "flair_bg_color": null,868 "flair_group_id": null,869 "trust_level": 1870 }871 ],872 "created_by": {873 "id": 20772,874 "username": "davidlee",875 "name": "",876 "avatar_template": "/user_avatar/discuss.pytorch.org/davidlee/{size}/12812_2.png"877 },878 "last_poster": {879 "id": 20772,880 "username": "davidlee",881 "name": "",882 "avatar_template": "/user_avatar/discuss.pytorch.org/davidlee/{size}/12812_2.png"883 }884 },885 "bookmarks": []886 },887 {888 "post_stream": {889 "posts": [890 {891 "id": 132623,892 "name": "",893 "username": "T_fighting",894 "avatar_template": "/letter_avatar_proxy/v4/letter/t/b9bd4f/{size}.png",895 "created_at": "2019-09-02T00:49:18.388Z",896 "cooked": "<p>The error always troubles me.I have refered to the official website for installing torch and torchvision by the package called pip .When I import torch,everything goes well;but when I import torchvision,the error always occurs。the screenshot shows below<br>\n<div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/2X/d/d7f3cb481a7eed0e7e4ba85aab6bc0390dea6128.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/d7f3cb481a7eed0e7e4ba85aab6bc0390dea6128\" title=\"Screenshot%20from%202019-09-02%2008-43-09\"><img src=\"https://discuss.pytorch.org/uploads/default/original/2X/d/d7f3cb481a7eed0e7e4ba85aab6bc0390dea6128.png\" alt=\"Screenshot%20from%202019-09-02%2008-43-09\" data-base62-sha1=\"uOoYUQuDVaY0koJaon5QDB7MvaM\" width=\"690\" height=\"429\" data-dominant-color=\"402036\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">Screenshot%20from%202019-09-02%2008-43-09</span><span class=\"informations\">803×500 91.4 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div></p>\n<p>The intresting thing is that everything is ok when I use tensorflow.<br>\nThe detail below is about TF2.0<br>\n‘’’<br>\n2019-09-02 08:39:51.517620: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcudart.so.10.0<br>\n2019-09-02 08:39:51.608876: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcublas.so.10.0<br>\n2019-09-02 08:39:51.636179: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcufft.so.10.0<br>\n2019-09-02 08:39:51.644710: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcurand.so.10.0<br>\n2019-09-02 08:39:51.705207: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcusolver.so.10.0<br>\n2019-09-02 08:39:51.742977: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcusparse.so.10.0<br>\n2019-09-02 08:39:51.861754: I tensorflow/stream_executor/platform/default/dso_loader.cc:42] Successfully opened dynamic library libcudnn.so.7<br>\n‘’’</p>\n<p>I hope who can help me to fix it.Thanks.</p>\n<p>ps:<br>\nOS:ubuntu18.04<br>\ncuda:10.0<br>\ntorch 1.2.0<br>\ntorchvision 0.3</p>",897 "post_number": 1,898 "post_type": 1,899 "posts_count": 5,900 "updated_at": "2019-09-02T00:50:58.795Z",901 "reply_count": 0,902 "reply_to_post_number": null,903 "quote_count": 0,904 "incoming_link_count": 456,905 "reads": 18,906 "readers_count": 17,907 "score": 2268.6,908 "yours": false,909 "topic_id": 54905,910 "topic_slug": "how-to-fix-the-error-importerror-libcudart-so-9-0",911 "display_username": "",912 "primary_group_name": null,913 "flair_name": null,914 "flair_url": null,915 "flair_bg_color": null,916 "flair_color": null,917 "flair_group_id": null,918 "badges_granted": [],919 "version": 1,920 "can_edit": false,921 "can_delete": false,922 "can_recover": false,923 "can_see_hidden_post": false,924 "can_wiki": false,925 "link_counts": [926 {927 "url": "https://discuss.pytorch.org/uploads/default/original/2X/d/d7f3cb481a7eed0e7e4ba85aab6bc0390dea6128.png",928 "internal": true,929 "reflection": false,930 "clicks": 0931 }932 ],933 "read": true,934 "user_title": null,935 "bookmarked": false,936 "actions_summary": [],937 "moderator": false,938 "admin": false,939 "staff": false,940 "user_id": 22161,941 "hidden": false,942 "trust_level": 0,943 "deleted_at": null,944 "user_deleted": false,945 "edit_reason": null,946 "can_view_edit_history": true,947 "wiki": false,948 "post_url": "/t/how-to-fix-the-error-importerror-libcudart-so-9-0/54905/1",949 "can_accept_answer": false,950 "can_unaccept_answer": false,951 "accepted_answer": false,952 "topic_accepted_answer": null,953 "can_vote": false954 },955 {956 "id": 132689,957 "name": "",958 "username": "ptrblck",959 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",960 "created_at": "2019-09-02T10:56:53.991Z",961 "cooked": "<p>Which binaries did you install (with which CUDA versions) or did you build from source?<br>\nCould you make sure to uninstall all older torchvision and PyTorch versions and reinstall them afterwards?<br>\nIt might also be a good idea to create a new virtual environment.</p>",962 "post_number": 2,963 "post_type": 1,964 "posts_count": 5,965 "updated_at": "2019-09-02T10:56:53.991Z",966 "reply_count": 0,967 "reply_to_post_number": null,968 "quote_count": 0,969 "incoming_link_count": 4,970 "reads": 18,971 "readers_count": 17,972 "score": 23.6,973 "yours": false,974 "topic_id": 54905,975 "topic_slug": "how-to-fix-the-error-importerror-libcudart-so-9-0",976 "display_username": "",977 "primary_group_name": null,978 "flair_name": null,979 "flair_url": null,980 "flair_bg_color": null,981 "flair_color": null,982 "flair_group_id": null,983 "badges_granted": [],984 "version": 1,985 "can_edit": false,986 "can_delete": false,987 "can_recover": false,988 "can_see_hidden_post": false,989 "can_wiki": false,990 "read": true,991 "user_title": "",992 "bookmarked": false,993 "actions_summary": [],994 "moderator": true,995 "admin": true,996 "staff": true,997 "user_id": 3534,998 "hidden": false,999 "trust_level": 2,1000 "deleted_at": null,1001 "user_deleted": false,1002 "edit_reason": null,1003 "can_view_edit_history": true,1004 "wiki": false,1005 "post_url": "/t/how-to-fix-the-error-importerror-libcudart-so-9-0/54905/2",1006 "can_accept_answer": false,1007 "can_unaccept_answer": false,1008 "accepted_answer": false,1009 "topic_accepted_answer": null1010 },1011 {1012 "id": 133729,1013 "name": "",1014 "username": "T_fighting",1015 "avatar_template": "/letter_avatar_proxy/v4/letter/t/b9bd4f/{size}.png",1016 "created_at": "2019-09-08T00:31:00.504Z",1017 "cooked": "<aside class=\"quote no-group\" data-username=\"ptrblck\" data-post=\"2\" data-topic=\"54905\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/user_avatar/discuss.pytorch.org/ptrblck/48/1823_2.png\" class=\"avatar\"> ptrblck:</div>\n<blockquote>\n<p>you insta</p>\n</blockquote>\n</aside>\n<p>Thanks for your answer! I have fixed it by uninstalling the current version(torchvision 0.4) and reinstalling the older torchvision(0.3). Now,everything goes well.</p>",1018 "post_number": 3,1019 "post_type": 1,1020 "posts_count": 5,1021 "updated_at": "2019-09-08T00:31:00.504Z",1022 "reply_count": 1,1023 "reply_to_post_number": 2,1024 "quote_count": 1,1025 "incoming_link_count": 5,1026 "reads": 13,1027 "readers_count": 12,1028 "score": 32.6,1029 "yours": false,1030 "topic_id": 54905,1031 "topic_slug": "how-to-fix-the-error-importerror-libcudart-so-9-0",1032 "display_username": "",1033 "primary_group_name": null,1034 "flair_name": null,1035 "flair_url": null,1036 "flair_bg_color": null,1037 "flair_color": null,1038 "flair_group_id": null,1039 "badges_granted": [],1040 "version": 1,1041 "can_edit": false,1042 "can_delete": false,1043 "can_recover": false,1044 "can_see_hidden_post": false,1045 "can_wiki": false,1046 "read": true,1047 "user_title": null,1048 "bookmarked": false,1049 "actions_summary": [],1050 "moderator": false,1051 "admin": false,1052 "staff": false,1053 "user_id": 22161,1054 "hidden": false,1055 "trust_level": 0,1056 "deleted_at": null,1057 "user_deleted": false,1058 "edit_reason": null,1059 "can_view_edit_history": true,1060 "wiki": false,1061 "post_url": "/t/how-to-fix-the-error-importerror-libcudart-so-9-0/54905/3",1062 "can_accept_answer": false,1063 "can_unaccept_answer": false,1064 "accepted_answer": false,1065 "topic_accepted_answer": null1066 },1067 {1068 "id": 133766,1069 "name": "",1070 "username": "ptrblck",1071 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1072 "created_at": "2019-09-08T11:06:50.743Z",1073 "cooked": "<p>Good to hear it’s working with the older version. Did you install <code>torchvision</code> 0.3 using the conda or pip binaries? If so, could you try to install the latest version using the same approach?</p>",1074 "post_number": 4,1075 "post_type": 1,1076 "posts_count": 5,1077 "updated_at": "2019-09-08T11:06:50.743Z",1078 "reply_count": 1,1079 "reply_to_post_number": 3,1080 "quote_count": 0,1081 "incoming_link_count": 1,1082 "reads": 11,1083 "readers_count": 10,1084 "score": 12.2,1085 "yours": false,1086 "topic_id": 54905,1087 "topic_slug": "how-to-fix-the-error-importerror-libcudart-so-9-0",1088 "display_username": "",1089 "primary_group_name": null,1090 "flair_name": null,1091 "flair_url": null,1092 "flair_bg_color": null,1093 "flair_color": null,1094 "flair_group_id": null,1095 "badges_granted": [],1096 "version": 1,1097 "can_edit": false,1098 "can_delete": false,1099 "can_recover": false,1100 "can_see_hidden_post": false,1101 "can_wiki": false,1102 "read": true,1103 "user_title": "",1104 "reply_to_user": {1105 "id": 22161,1106 "username": "T_fighting",1107 "name": "",1108 "avatar_template": "/letter_avatar_proxy/v4/letter/t/b9bd4f/{size}.png"1109 },1110 "bookmarked": false,1111 "actions_summary": [],1112 "moderator": true,1113 "admin": true,1114 "staff": true,1115 "user_id": 3534,1116 "hidden": false,1117 "trust_level": 2,1118 "deleted_at": null,1119 "user_deleted": false,1120 "edit_reason": null,1121 "can_view_edit_history": true,1122 "wiki": false,1123 "post_url": "/t/how-to-fix-the-error-importerror-libcudart-so-9-0/54905/4",1124 "can_accept_answer": false,1125 "can_unaccept_answer": false,1126 "accepted_answer": false,1127 "topic_accepted_answer": null1128 },1129 {1130 "id": 133871,1131 "name": "",1132 "username": "T_fighting",1133 "avatar_template": "/letter_avatar_proxy/v4/letter/t/b9bd4f/{size}.png",1134 "created_at": "2019-09-09T07:55:09.671Z",1135 "cooked": "<p>Ok! I will try it! Enjoy pytorch!</p>",1136 "post_number": 5,1137 "post_type": 1,1138 "posts_count": 5,1139 "updated_at": "2019-09-09T07:55:09.671Z",1140 "reply_count": 0,1141 "reply_to_post_number": 4,1142 "quote_count": 0,1143 "incoming_link_count": 2,1144 "reads": 9,1145 "readers_count": 8,1146 "score": 11.8,1147 "yours": false,1148 "topic_id": 54905,1149 "topic_slug": "how-to-fix-the-error-importerror-libcudart-so-9-0",1150 "display_username": "",1151 "primary_group_name": null,1152 "flair_name": null,1153 "flair_url": null,1154 "flair_bg_color": null,1155 "flair_color": null,1156 "flair_group_id": null,1157 "badges_granted": [],1158 "version": 1,1159 "can_edit": false,1160 "can_delete": false,1161 "can_recover": false,1162 "can_see_hidden_post": false,1163 "can_wiki": false,1164 "read": true,1165 "user_title": null,1166 "reply_to_user": {1167 "id": 3534,1168 "username": "ptrblck",1169 "name": "",1170 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"1171 },1172 "bookmarked": false,1173 "actions_summary": [],1174 "moderator": false,1175 "admin": false,1176 "staff": false,1177 "user_id": 22161,1178 "hidden": false,1179 "trust_level": 0,1180 "deleted_at": null,1181 "user_deleted": false,1182 "edit_reason": null,1183 "can_view_edit_history": true,1184 "wiki": false,1185 "post_url": "/t/how-to-fix-the-error-importerror-libcudart-so-9-0/54905/5",1186 "can_accept_answer": false,1187 "can_unaccept_answer": false,1188 "accepted_answer": false,1189 "topic_accepted_answer": null1190 }1191 ],1192 "stream": [1193 132623,1194 132689,1195 133729,1196 133766,1197 1338711198 ]1199 },1200 "timeline_lookup": [