Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 264210,7 "name": "Samuel Bachorik",8 "username": "Samuel_Bachorik",9 "avatar_template": "/user_avatar/discuss.pytorch.org/samuel_bachorik/{size}/56232_2.png",10 "created_at": "2021-02-17T17:41:48.341Z",11 "cooked": "<p>Hello, i am using two computers for training, one have linux and second windows. These PCs have same HW. When i try to run same training on windows witch batch 4 iam getting CUDA out of memory. On Linux i can have batch 16 and it works fine. To me it seems windows is using too much of GPU memory, is there any way to fix this ? Decrease GPU memory usage by windows ?</p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 2,15 "updated_at": "2021-02-17T17:41:48.341Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 45,20 "reads": 8,21 "readers_count": 7,22 "score": 226.6,23 "yours": false,24 "topic_id": 112180,25 "topic_slug": "gpu-memory-usage-windows",26 "display_username": "Samuel Bachorik",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 1,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "read": true,41 "user_title": "",42 "bookmarked": false,43 "actions_summary": [],44 "moderator": false,45 "admin": false,46 "staff": false,47 "user_id": 41434,48 "hidden": false,49 "trust_level": 2,50 "deleted_at": null,51 "user_deleted": false,52 "edit_reason": null,53 "can_view_edit_history": true,54 "wiki": false,55 "post_url": "/t/gpu-memory-usage-windows/112180/1",56 "can_accept_answer": false,57 "can_unaccept_answer": false,58 "accepted_answer": false,59 "topic_accepted_answer": null,60 "can_vote": false61 },62 {63 "id": 264287,64 "name": "",65 "username": "ptrblck",66 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",67 "created_at": "2021-02-18T05:35:20.199Z",68 "cooked": "<p>Are the GPUs completely empty in both setups or are you using e.g. a display or other applications in the Windows machine?<br>\nIf they are empty (check it via <code>nvidia-smi</code>), are you seeing the larger memory usage in all models on the Windows system?</p>",69 "post_number": 2,70 "post_type": 1,71 "posts_count": 2,72 "updated_at": "2021-02-18T05:35:20.199Z",73 "reply_count": 0,74 "reply_to_post_number": null,75 "quote_count": 0,76 "incoming_link_count": 0,77 "reads": 5,78 "readers_count": 4,79 "score": 16.0,80 "yours": false,81 "topic_id": 112180,82 "topic_slug": "gpu-memory-usage-windows",83 "display_username": "",84 "primary_group_name": null,85 "flair_name": null,86 "flair_url": null,87 "flair_bg_color": null,88 "flair_color": null,89 "flair_group_id": null,90 "badges_granted": [],91 "version": 1,92 "can_edit": false,93 "can_delete": false,94 "can_recover": false,95 "can_see_hidden_post": false,96 "can_wiki": false,97 "read": true,98 "user_title": "",99 "bookmarked": false,100 "actions_summary": [101 {102 "id": 2,103 "count": 1104 }105 ],106 "moderator": true,107 "admin": true,108 "staff": true,109 "user_id": 3534,110 "hidden": false,111 "trust_level": 2,112 "deleted_at": null,113 "user_deleted": false,114 "edit_reason": null,115 "can_view_edit_history": true,116 "wiki": false,117 "post_url": "/t/gpu-memory-usage-windows/112180/2",118 "can_accept_answer": false,119 "can_unaccept_answer": false,120 "accepted_answer": false,121 "topic_accepted_answer": null122 }123 ],124 "stream": [125 264210,126 264287127 ]128 },129 "timeline_lookup": [130 [131 1,132 1711133 ]134 ],135 "suggested_topics": [136 {137 "fancy_title": "Asynchronous execution multiple GPUs",138 "id": 212448,139 "title": "Asynchronous execution multiple GPUs",140 "slug": "asynchronous-execution-multiple-gpus",141 "posts_count": 3,142 "reply_count": 1,143 "highest_post_number": 3,144 "image_url": null,145 "created_at": "2024-11-02T12:59:53.714Z",146 "last_posted_at": "2024-11-05T13:22:01.243Z",147 "bumped": true,148 "bumped_at": "2024-11-05T13:22:01.243Z",149 "archetype": "regular",150 "unseen": false,151 "pinned": false,152 "unpinned": null,153 "visible": true,154 "closed": false,155 "archived": false,156 "bookmarked": null,157 "liked": null,158 "tags_descriptions": {},159 "like_count": 0,160 "views": 195,161 "category_id": 1,162 "featured_link": null,163 "has_accepted_answer": false,164 "posters": [165 {166 "extras": "latest",167 "description": "Original Poster, Most Recent Poster",168 "user": {169 "id": 71569,170 "username": "Manvel_Petrosyan",171 "name": "Manvel Petrosyan",172 "avatar_template": "/user_avatar/discuss.pytorch.org/manvel_petrosyan/{size}/66101_2.png",173 "trust_level": 1174 }175 },176 {177 "extras": null,178 "description": "Frequent Poster",179 "user": {180 "id": 3534,181 "username": "ptrblck",182 "name": "",183 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",184 "admin": true,185 "moderator": true,186 "trust_level": 2187 }188 }189 ]190 },191 {192 "fancy_title": "Seeking advice on computing backprop step by chunks",193 "id": 212924,194 "title": "Seeking advice on computing backprop step by chunks",195 "slug": "seeking-advice-on-computing-backprop-step-by-chunks",196 "posts_count": 2,197 "reply_count": 0,198 "highest_post_number": 2,199 "image_url": null,200 "created_at": "2024-11-13T12:28:34.725Z",201 "last_posted_at": "2024-11-14T14:09:04.532Z",202 "bumped": true,203 "bumped_at": "2024-11-14T14:09:04.532Z",204 "archetype": "regular",205 "unseen": false,206 "pinned": false,207 "unpinned": null,208 "visible": true,209 "closed": false,210 "archived": false,211 "bookmarked": null,212 "liked": null,213 "tags_descriptions": {},214 "like_count": 0,215 "views": 36,216 "category_id": 1,217 "featured_link": null,218 "has_accepted_answer": true,219 "posters": [220 {221 "extras": "latest single",222 "description": "Original Poster, Most Recent Poster, Accepted Answer",223 "user": {224 "id": 80885,225 "username": "meditans",226 "name": "",227 "avatar_template": "/user_avatar/discuss.pytorch.org/meditans/{size}/73976_2.png",228 "trust_level": 1229 }230 }231 ]232 },233 {234 "fancy_title": "Turning large matrix into a lower triangular matrix",235 "id": 214469,236 "title": "Turning large matrix into a lower triangular matrix",237 "slug": "turning-large-matrix-into-a-lower-triangular-matrix",238 "posts_count": 2,239 "reply_count": 0,240 "highest_post_number": 2,241 "image_url": null,242 "created_at": "2024-12-20T19:46:58.211Z",243 "last_posted_at": "2024-12-20T23:20:22.186Z",244 "bumped": true,245 "bumped_at": "2024-12-20T23:20:22.186Z",246 "archetype": "regular",247 "unseen": false,248 "pinned": false,249 "unpinned": null,250 "visible": true,251 "closed": false,252 "archived": false,253 "bookmarked": null,254 "liked": null,255 "tags_descriptions": {},256 "like_count": 1,257 "views": 24,258 "category_id": 1,259 "featured_link": null,260 "has_accepted_answer": true,261 "posters": [262 {263 "extras": null,264 "description": "Original Poster",265 "user": {266 "id": 68506,267 "username": "alexj",268 "name": "Alex Jouravlev",269 "avatar_template": "/user_avatar/discuss.pytorch.org/alexj/{size}/70445_2.png",270 "trust_level": 1271 }272 },273 {274 "extras": "latest",275 "description": "Most Recent Poster, Accepted Answer",276 "user": {277 "id": 41396,278 "username": "soulitzer",279 "name": "",280 "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",281 "trust_level": 2282 }283 }284 ]285 },286 {287 "fancy_title": "multiprocessing.get_context(‘spawn’).Pool creates too many processes",288 "id": 215951,289 "title": "multiprocessing.get_context('spawn').Pool creates too many processes",290 "slug": "multiprocessing-get-context-spawn-pool-creates-too-many-processes",291 "posts_count": 2,292 "reply_count": 0,293 "highest_post_number": 2,294 "image_url": null,295 "created_at": "2025-01-27T18:04:44.483Z",296 "last_posted_at": "2025-01-27T18:49:00.462Z",297 "bumped": true,298 "bumped_at": "2025-01-27T18:49:00.462Z",299 "archetype": "regular",300 "unseen": false,301 "pinned": false,302 "unpinned": null,303 "visible": true,304 "closed": false,305 "archived": false,306 "bookmarked": null,307 "liked": null,308 "tags_descriptions": {},309 "like_count": 0,310 "views": 200,311 "category_id": 1,312 "featured_link": null,313 "has_accepted_answer": true,314 "posters": [315 {316 "extras": "latest single",317 "description": "Original Poster, Most Recent Poster, Accepted Answer",318 "user": {319 "id": 53378,320 "username": "foshea",321 "name": "Finn",322 "avatar_template": "/letter_avatar_proxy/v4/letter/f/a6a055/{size}.png",323 "trust_level": 1324 }325 }326 ]327 },328 {329 "fancy_title": "torch.export.ExportedProgram",330 "id": 217502,331 "title": "torch.export.ExportedProgram",332 "slug": "torch-export-exportedprogram",333 "posts_count": 1,334 "reply_count": 0,335 "highest_post_number": 1,336 "image_url": null,337 "created_at": "2025-03-06T04:54:17.482Z",338 "last_posted_at": "2025-03-06T04:54:17.519Z",339 "bumped": true,340 "bumped_at": "2025-03-06T04:54:17.519Z",341 "archetype": "regular",342 "unseen": false,343 "pinned": false,344 "unpinned": null,345 "visible": true,346 "closed": false,347 "archived": false,348 "bookmarked": null,349 "liked": null,350 "tags_descriptions": {},351 "like_count": 0,352 "views": 58,353 "category_id": 1,354 "featured_link": null,355 "has_accepted_answer": false,356 "posters": [357 {358 "extras": "latest single",359 "description": "Original Poster, Most Recent Poster",360 "user": {361 "id": 83097,362 "username": "qianfenchi",363 "name": "qianfenchi",364 "avatar_template": "/user_avatar/discuss.pytorch.org/qianfenchi/{size}/76008_2.png",365 "trust_level": 0366 }367 }368 ]369 }370 ],371 "tags_descriptions": {},372 "fancy_title": "GPU Memory usage, Windows",373 "id": 112180,374 "title": "GPU Memory usage, Windows",375 "posts_count": 2,376 "created_at": "2021-02-17T17:41:48.270Z",377 "views": 408,378 "reply_count": 0,379 "like_count": 1,380 "last_posted_at": "2021-02-18T05:35:20.199Z",381 "visible": true,382 "closed": false,383 "archived": false,384 "has_summary": false,385 "archetype": "regular",386 "slug": "gpu-memory-usage-windows",387 "category_id": 1,388 "word_count": 119,389 "deleted_at": null,390 "user_id": 41434,391 "featured_link": null,392 "pinned_globally": false,393 "pinned_at": null,394 "pinned_until": null,395 "image_url": null,396 "slow_mode_seconds": 0,397 "draft": null,398 "draft_key": "topic_112180",399 "draft_sequence": null,400 "unpinned": null,401 "pinned": false,402 "current_post_number": 1,403 "highest_post_number": 2,404 "deleted_by": null,405 "actions_summary": [406 {407 "id": 4,408 "count": 0,409 "hidden": false,410 "can_act": false411 },412 {413 "id": 8,414 "count": 0,415 "hidden": false,416 "can_act": false417 },418 {419 "id": 10,420 "count": 0,421 "hidden": false,422 "can_act": false423 },424 {425 "id": 7,426 "count": 0,427 "hidden": false,428 "can_act": false429 }430 ],431 "chunk_size": 20,432 "bookmarked": false,433 "topic_timer": null,434 "message_bus_last_id": 0,435 "participant_count": 2,436 "show_read_indicator": false,437 "thumbnails": null,438 "slow_mode_enabled_until": null,439 "can_vote": false,440 "vote_count": 0,441 "user_voted": false,442 "discourse_zendesk_plugin_zendesk_id": null,443 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",444 "details": {445 "can_edit": false,446 "notification_level": 1,447 "participants": [448 {449 "id": 3534,450 "username": "ptrblck",451 "name": "",452 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",453 "post_count": 1,454 "primary_group_name": null,455 "flair_name": null,456 "flair_url": null,457 "flair_color": null,458 "flair_bg_color": null,459 "flair_group_id": null,460 "admin": true,461 "moderator": true,462 "trust_level": 2463 },464 {465 "id": 41434,466 "username": "Samuel_Bachorik",467 "name": "Samuel Bachorik",468 "avatar_template": "/user_avatar/discuss.pytorch.org/samuel_bachorik/{size}/56232_2.png",469 "post_count": 1,470 "primary_group_name": null,471 "flair_name": null,472 "flair_url": null,473 "flair_color": null,474 "flair_bg_color": null,475 "flair_group_id": null,476 "trust_level": 2477 }478 ],479 "created_by": {480 "id": 41434,481 "username": "Samuel_Bachorik",482 "name": "Samuel Bachorik",483 "avatar_template": "/user_avatar/discuss.pytorch.org/samuel_bachorik/{size}/56232_2.png"484 },485 "last_poster": {486 "id": 3534,487 "username": "ptrblck",488 "name": "",489 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"490 }491 },492 "bookmarks": []493 },494 {495 "post_stream": {496 "posts": [497 {498 "id": 264275,499 "name": "Lanxin HE",500 "username": "LanxinHE",501 "avatar_template": "/letter_avatar_proxy/v4/letter/l/6de8d8/{size}.png",502 "created_at": "2021-02-18T02:42:10.337Z",503 "cooked": "<p>Hi, everyone, Here is a bug needing your help.<br>\nI am training my own model using tensorboardx to record loss. At the beginning everything goes well, and tensorboard shows the normal loss pic. However, when the epoch reaches 191(last time it was 110+), the program got stuck. It did NOT throw any error but never went ahead:<br>\n<img src=\"https://discuss.pytorch.org/uploads/default/original/3X/2/f/2f59ee5e4f6ef8da004d4e51c0f1b195da6c916c.png\" alt=\"image\" data-base62-sha1=\"6KT7ihWE24vmPVJEa1eLXdoQzGQ\" width=\"352\" height=\"212\"><br>\nmaking a breakpoint at main_file does not help. When I pause the program, after a slightly long time, I landed in connection.py at:<div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/1620bb46006b7cc59d1f6472c241205694156100\" title=\"image\"><img src=\"https://discuss.pytorch.org/uploads/default/optimized/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100_2_690x371.png\" alt=\"image\" data-base62-sha1=\"39KD8T0GvWDDolESj50Ir0g09pe\" width=\"690\" height=\"371\" srcset=\"https://discuss.pytorch.org/uploads/default/optimized/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100_2_690x371.png, https://discuss.pytorch.org/uploads/default/original/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100.png 1.5x, https://discuss.pytorch.org/uploads/default/original/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100.png 2x\" data-dominant-color=\"2F3235\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">image</span><span class=\"informations\">968×521 47.2 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div><br>\nstepping out continuously gets in event_file_writer.py:<br>\n<div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/052340e2daa75f3f9b219f7432502d60458e12ba\" title=\"image\"><img src=\"https://discuss.pytorch.org/uploads/default/optimized/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba_2_690x326.png\" alt=\"image\" data-base62-sha1=\"JrUT2IiiHjqyMnidAEQNaZqk1I\" width=\"690\" height=\"326\" srcset=\"https://discuss.pytorch.org/uploads/default/optimized/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba_2_690x326.png, https://discuss.pytorch.org/uploads/default/optimized/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba_2_1035x489.png 1.5x, https://discuss.pytorch.org/uploads/default/original/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba.png 2x\" data-dominant-color=\"2F2E2E\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">image</span><span class=\"informations\">1055×499 45.8 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div><br>\nHas anyone ever encountered this? or this is the bug in tensorboardx caused by a non-matched version?</p>",504 "post_number": 1,505 "post_type": 1,506 "posts_count": 1,507 "updated_at": "2021-02-18T02:44:19.393Z",508 "reply_count": 0,509 "reply_to_post_number": null,510 "quote_count": 0,511 "incoming_link_count": 81,512 "reads": 4,513 "readers_count": 3,514 "score": 415.8,515 "yours": false,516 "topic_id": 112209,517 "topic_slug": "got-stuck-at-tensorboardx-event-file-writer-py",518 "display_username": "Lanxin HE",519 "primary_group_name": null,520 "flair_name": null,521 "flair_url": null,522 "flair_bg_color": null,523 "flair_color": null,524 "flair_group_id": null,525 "badges_granted": [],526 "version": 1,527 "can_edit": false,528 "can_delete": false,529 "can_recover": false,530 "can_see_hidden_post": false,531 "can_wiki": false,532 "link_counts": [533 {534 "url": "https://discuss.pytorch.org/uploads/default/original/3X/1/6/1620bb46006b7cc59d1f6472c241205694156100.png",535 "internal": true,536 "reflection": false,537 "clicks": 0538 },539 {540 "url": "https://discuss.pytorch.org/uploads/default/original/3X/0/5/052340e2daa75f3f9b219f7432502d60458e12ba.png",541 "internal": true,542 "reflection": false,543 "clicks": 0544 }545 ],546 "read": true,547 "user_title": null,548 "bookmarked": false,549 "actions_summary": [550 {551 "id": 2,552 "count": 1553 }554 ],555 "moderator": false,556 "admin": false,557 "staff": false,558 "user_id": 41348,559 "hidden": false,560 "trust_level": 1,561 "deleted_at": null,562 "user_deleted": false,563 "edit_reason": null,564 "can_view_edit_history": true,565 "wiki": false,566 "post_url": "/t/got-stuck-at-tensorboardx-event-file-writer-py/112209/1",567 "can_accept_answer": false,568 "can_unaccept_answer": false,569 "accepted_answer": false,570 "topic_accepted_answer": null,571 "can_vote": false572 }573 ],574 "stream": [575 264275576 ]577 },578 "timeline_lookup": [579 [580 1,581 1711582 ]583 ],584 "suggested_topics": [585 {586 "fancy_title": "Tensorboard add_graph (actually jit.trace) does not work with None inputs",587 "id": 219863,588 "title": "Tensorboard add_graph (actually jit.trace) does not work with None inputs",589 "slug": "tensorboard-add-graph-actually-jit-trace-does-not-work-with-none-inputs",590 "posts_count": 2,591 "reply_count": 0,592 "highest_post_number": 2,593 "image_url": null,594 "created_at": "2025-05-08T09:44:13.923Z",595 "last_posted_at": "2025-05-08T09:52:55.880Z",596 "bumped": true,597 "bumped_at": "2025-05-08T10:04:01.686Z",598 "archetype": "regular",599 "unseen": false,600 "pinned": false,601 "unpinned": null,602 "visible": true,603 "closed": false,604 "archived": false,605 "bookmarked": null,606 "liked": null,607 "tags_descriptions": {},608 "like_count": 0,609 "views": 65,610 "category_id": 28,611 "featured_link": null,612 "has_accepted_answer": false,613 "posters": [614 {615 "extras": "latest single",616 "description": "Original Poster, Most Recent Poster",617 "user": {618 "id": 75257,619 "username": "turbotimon",620 "name": "Timon Erhart",621 "avatar_template": "/user_avatar/discuss.pytorch.org/turbotimon/{size}/69503_2.png",622 "trust_level": 1623 }624 }625 ]626 },627 {628 "fancy_title": "No download option for histogram visualizations in TensorBoard (when using torch-pruning)",629 "id": 222479,630 "title": "No download option for histogram visualizations in TensorBoard (when using torch-pruning)",631 "slug": "no-download-option-for-histogram-visualizations-in-tensorboard-when-using-torch-pruning",632 "posts_count": 1,633 "reply_count": 0,634 "highest_post_number": 1,635 "image_url": null,636 "created_at": "2025-08-19T12:08:54.212Z",637 "last_posted_at": "2025-08-19T12:08:54.266Z",638 "bumped": true,639 "bumped_at": "2025-08-19T12:08:54.266Z",640 "archetype": "regular",641 "unseen": false,642 "pinned": false,643 "unpinned": null,644 "visible": true,645 "closed": false,646 "archived": false,647 "bookmarked": null,648 "liked": null,649 "tags_descriptions": {},650 "like_count": 0,651 "views": 26,652 "category_id": 28,653 "featured_link": null,654 "has_accepted_answer": false,655 "posters": [656 {657 "extras": "latest single",658 "description": "Original Poster, Most Recent Poster",659 "user": {660 "id": 85553,661 "username": "SadMoon",662 "name": "",663 "avatar_template": "/letter_avatar_proxy/v4/letter/s/ac91a4/{size}.png",664 "trust_level": 0665 }666 }667 ]668 },669 {670 "fancy_title": "How am i supposed to code an axonal delay in my SNN?",671 "id": 223163,672 "title": "How am i supposed to code an axonal delay in my SNN?",673 "slug": "how-am-i-supposed-to-code-an-axonal-delay-in-my-snn",674 "posts_count": 1,675 "reply_count": 0,676 "highest_post_number": 1,677 "image_url": null,678 "created_at": "2025-09-18T07:15:54.444Z",679 "last_posted_at": "2025-09-18T07:15:54.504Z",680 "bumped": true,681 "bumped_at": "2025-09-18T09:55:48.325Z",682 "archetype": "regular",683 "unseen": false,684 "pinned": false,685 "unpinned": null,686 "visible": true,687 "closed": false,688 "archived": false,689 "bookmarked": null,690 "liked": null,691 "tags_descriptions": {},692 "like_count": 0,693 "views": 23,694 "category_id": 28,695 "featured_link": null,696 "has_accepted_answer": false,697 "posters": [698 {699 "extras": "latest single",700 "description": "Original Poster, Most Recent Poster",701 "user": {702 "id": 85898,703 "username": "Photon1",704 "name": "Photon",705 "avatar_template": "/user_avatar/discuss.pytorch.org/photon1/{size}/77068_2.png",706 "trust_level": 0707 }708 }709 ]710 },711 {712 "fancy_title": "How to embed pytorch profiler to verl?",713 "id": 218916,714 "title": "How to embed pytorch profiler to verl?",715 "slug": "how-to-embed-pytorch-profiler-to-verl",716 "posts_count": 1,717 "reply_count": 0,718 "highest_post_number": 1,719 "image_url": null,720 "created_at": "2025-04-09T18:38:57.973Z",721 "last_posted_at": "2025-04-09T18:38:58.021Z",722 "bumped": true,723 "bumped_at": "2025-04-09T18:38:58.021Z",724 "archetype": "regular",725 "unseen": false,726 "pinned": false,727 "unpinned": null,728 "visible": true,729 "closed": false,730 "archived": false,731 "bookmarked": null,732 "liked": null,733 "tags_descriptions": {},734 "like_count": 0,735 "views": 106,736 "category_id": 28,737 "featured_link": null,738 "has_accepted_answer": false,739 "posters": [740 {741 "extras": "latest single",742 "description": "Original Poster, Most Recent Poster",743 "user": {744 "id": 83728,745 "username": "gogineni_kailash",746 "name": "gogineni kailashnath",747 "avatar_template": "/user_avatar/discuss.pytorch.org/gogineni_kailash/{size}/76552_2.png",748 "trust_level": 1749 }750 }751 ]752 },753 {754 "fancy_title": "Call backward() twice does not incur errors",755 "id": 221127,756 "title": "Call backward() twice does not incur errors",757 "slug": "call-backward-twice-does-not-incur-errors",758 "posts_count": 2,759 "reply_count": 0,760 "highest_post_number": 2,761 "image_url": "https://discuss.pytorch.org/uploads/default/optimized/3X/e/2/e21a24d19d4e37445c62930c66f5a278967aacb4_2_1023x185.png",762 "created_at": "2025-06-28T08:18:46.513Z",763 "last_posted_at": "2025-06-28T15:57:00.993Z",764 "bumped": true,765 "bumped_at": "2025-06-28T15:57:00.993Z",766 "archetype": "regular",767 "unseen": false,768 "pinned": false,769 "unpinned": null,770 "visible": true,771 "closed": false,772 "archived": false,773 "bookmarked": null,774 "liked": null,775 "tags_descriptions": {},776 "like_count": 1,777 "views": 38,778 "category_id": 1,779 "featured_link": null,780 "has_accepted_answer": true,781 "posters": [782 {783 "extras": null,784 "description": "Original Poster",785 "user": {786 "id": 49342,787 "username": "bingqing_liu",788 "name": "bingqing liu",789 "avatar_template": "/user_avatar/discuss.pytorch.org/bingqing_liu/{size}/31455_2.png",790 "trust_level": 1791 }792 },793 {794 "extras": "latest",795 "description": "Most Recent Poster, Accepted Answer",796 "user": {797 "id": 41396,798 "username": "soulitzer",799 "name": "",800 "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",801 "trust_level": 2802 }803 }804 ]805 }806 ],807 "tags_descriptions": {},808 "fancy_title": "Got stuck at tensorboardx event_file_writer.py",809 "id": 112209,810 "title": "Got stuck at tensorboardx event_file_writer.py",811 "posts_count": 1,812 "created_at": "2021-02-18T02:42:10.272Z",813 "views": 992,814 "reply_count": 0,815 "like_count": 1,816 "last_posted_at": "2021-02-18T02:42:10.337Z",817 "visible": true,818 "closed": false,819 "archived": false,820 "has_summary": false,821 "archetype": "regular",822 "slug": "got-stuck-at-tensorboardx-event-file-writer-py",823 "category_id": 28,824 "word_count": 122,825 "deleted_at": null,826 "user_id": 41348,827 "featured_link": null,828 "pinned_globally": false,829 "pinned_at": null,830 "pinned_until": null,831 "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/2/f/2f59ee5e4f6ef8da004d4e51c0f1b195da6c916c.png",832 "slow_mode_seconds": 0,833 "draft": null,834 "draft_key": "topic_112209",835 "draft_sequence": null,836 "unpinned": null,837 "pinned": false,838 "current_post_number": 1,839 "highest_post_number": 1,840 "deleted_by": null,841 "actions_summary": [842 {843 "id": 4,844 "count": 0,845 "hidden": false,846 "can_act": false847 },848 {849 "id": 8,850 "count": 0,851 "hidden": false,852 "can_act": false853 },854 {855 "id": 10,856 "count": 0,857 "hidden": false,858 "can_act": false859 },860 {861 "id": 7,862 "count": 0,863 "hidden": false,864 "can_act": false865 }866 ],867 "chunk_size": 20,868 "bookmarked": false,869 "topic_timer": null,870 "message_bus_last_id": 0,871 "participant_count": 1,872 "show_read_indicator": false,873 "thumbnails": [874 {875 "max_width": null,876 "max_height": null,877 "width": 352,878 "height": 212,879 "url": "https://discuss.pytorch.org/uploads/default/original/3X/2/f/2f59ee5e4f6ef8da004d4e51c0f1b195da6c916c.png"880 }881 ],882 "slow_mode_enabled_until": null,883 "can_vote": false,884 "vote_count": 0,885 "user_voted": false,886 "discourse_zendesk_plugin_zendesk_id": null,887 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",888 "details": {889 "can_edit": false,890 "notification_level": 1,891 "participants": [892 {893 "id": 41348,894 "username": "LanxinHE",895 "name": "Lanxin HE",896 "avatar_template": "/letter_avatar_proxy/v4/letter/l/6de8d8/{size}.png",897 "post_count": 1,898 "primary_group_name": null,899 "flair_name": null,900 "flair_url": null,901 "flair_color": null,902 "flair_bg_color": null,903 "flair_group_id": null,904 "trust_level": 1905 }906 ],907 "created_by": {908 "id": 41348,909 "username": "LanxinHE",910 "name": "Lanxin HE",911 "avatar_template": "/letter_avatar_proxy/v4/letter/l/6de8d8/{size}.png"912 },913 "last_poster": {914 "id": 41348,915 "username": "LanxinHE",916 "name": "Lanxin HE",917 "avatar_template": "/letter_avatar_proxy/v4/letter/l/6de8d8/{size}.png"918 }919 },920 "bookmarks": []921 },922 {923 "post_stream": {924 "posts": [925 {926 "id": 263803,927 "name": "Julianolm",928 "username": "julianolm",929 "avatar_template": "/user_avatar/discuss.pytorch.org/julianolm/{size}/35135_2.png",930 "created_at": "2021-02-15T22:23:32.883Z",931 "cooked": "<p>I’m having some trouble implementing a Local Binary Convolutional Neural Network (LBCNN) in Pytorch.</p>\n<p>Such model uses fixed convolutional binary filters, they are “randomly” set during the instantiation of the net and keep the same ever after. The only parameter updates are over linear weights that combine the feature maps generated by the filters (this operation is implemented with 1x1 convolutions). The network is then expected to have a small number of learnable parameters and, hence, to be easy to train. But happens that I am receiving the following message when trying to train it:</p>\n<blockquote>\n<p>RuntimeError: CUDA out of memory. Tried to allocate 128.00 MiB (GPU 0; 15.90 GiB total capacity; 13.75 GiB already allocated; 7.75 MiB free; 15.09 GiB reserved in total by PyTorch)</p>\n</blockquote>\n<p>Here is the code for network:</p>\n<pre><code>class LBCBlock(nn.Module):\n def __init__(self, n_channels=384, n_kernels=512, sparsity=0.1):\n super().__init__()\n self.n_channels = n_channels\n self.n_kernels = n_kernels\n\n self.conv_filter = nn.Conv2d(n_channels, n_kernels, kernel_size=3, padding=1, bias=False)\n kernels = torch.tensor(new_kernel(n_channels, n_kernels, sparsity)).type('torch.FloatTensor')\n kernels.requires_grad_()\n self.conv_filter.weight = nn.Parameter(kernels)\n self.conv_filter.weight.detach_()\n\n self.weighted_sum = nn.Conv2d(n_kernels, n_channels, kernel_size=1)\n\n self.shortcut = nn.Sequential()\n\n def forward(self, x):\n out = torch.relu(self.conv_filter(x))\n out = self.weighted_sum(out)\n out += self.shortcut(x)\n\n return out\n\nimg_width = 32\nimg_height = 32\nclass NetLBC(nn.Module):\n def __init__(self, lbc_filters=512, n_channels=384, n_blocks=10, sparsity=0.1):\n super().__init__()\n self.n_channels = n_channels\n self.n_blocks = n_blocks\n\n self.conv1 = nn.Conv2d(3, n_channels, kernel_size=3, padding=1)\n self.lbc_blocks = nn.Sequential(*( [LBCBlock(n_channels, lbc_filters, sparsity) for i in range(n_blocks)]))\n\n self.fc1 = nn.Linear(self.n_channels * (img_height//2) * (img_width//2), 384)\n self.fc2 = nn.Linear(384, 10)\n\n def forward(self, x):\n out = self.conv1(x)\n out = self.lbc_blocks(out)\n out = torch.nn.functional.max_pool2d(out, 2)\n \n out = out.view(-1, self.n_channels * (img_height//2) * (img_width//2))\n\n out = torch.relu(self.fc1(out))\n out = self.fc2(out)\n\n return out\n\nmodel = NetLBC(lbc_filters=512, n_channels=384, n_blocks=50, sparsity=0.1).to(device=device)\n</code></pre>",932 "post_number": 1,933 "post_type": 1,934 "posts_count": 4,935 "updated_at": "2021-02-15T22:23:32.883Z",936 "reply_count": 1,937 "reply_to_post_number": null,938 "quote_count": 0,939 "incoming_link_count": 78,940 "reads": 5,941 "readers_count": 4,942 "score": 396.0,943 "yours": false,944 "topic_id": 111980,945 "topic_slug": "how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn",946 "display_username": "Julianolm",947 "primary_group_name": null,948 "flair_name": null,949 "flair_url": null,950 "flair_bg_color": null,951 "flair_color": null,952 "flair_group_id": null,953 "badges_granted": [],954 "version": 1,955 "can_edit": false,956 "can_delete": false,957 "can_recover": false,958 "can_see_hidden_post": false,959 "can_wiki": false,960 "read": true,961 "user_title": null,962 "bookmarked": false,963 "actions_summary": [],964 "moderator": false,965 "admin": false,966 "staff": false,967 "user_id": 42138,968 "hidden": false,969 "trust_level": 1,970 "deleted_at": null,971 "user_deleted": false,972 "edit_reason": null,973 "can_view_edit_history": true,974 "wiki": false,975 "post_url": "/t/how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn/111980/1",976 "can_accept_answer": false,977 "can_unaccept_answer": false,978 "accepted_answer": false,979 "topic_accepted_answer": null,980 "can_vote": false981 },982 {983 "id": 263815,984 "name": "",985 "username": "ptrblck",986 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",987 "created_at": "2021-02-15T23:48:32.738Z",988 "cooked": "<aside class=\"quote no-group\" data-username=\"julianolm\" data-post=\"1\" data-topic=\"111980\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/user_avatar/discuss.pytorch.org/julianolm/48/35135_2.png\" class=\"avatar\"> julianolm:</div>\n<blockquote>\n<p>Such model uses fixed convolutional binary filters, they are “randomly” set during the instantiation of the net and keep the same ever after. The only parameter updates are over linear weights that combine the feature maps generated by the filters (this operation is implemented with 1x1 convolutions).</p>\n</blockquote>\n</aside>\n<p>Assuming the <code>1x1</code> convolutions are implemented via the <code>self.fc1</code> layer in <code>NetLBC</code>, I understand that all preceding layers should not be updated.<br>\nIf that’s the case, you could wrap the first part of the model into a <code>with torch.no_grad()</code> block to avoid storing intermediate activations, which would be needed to compute the gradients.<br>\nAlso, lowering the batch size would decrease the memory usage and should help.</p>",989 "post_number": 2,990 "post_type": 1,991 "posts_count": 4,992 "updated_at": "2021-02-15T23:48:32.738Z",993 "reply_count": 1,994 "reply_to_post_number": null,995 "quote_count": 1,996 "incoming_link_count": 5,997 "reads": 5,998 "readers_count": 4,999 "score": 31.0,1000 "yours": false,1001 "topic_id": 111980,1002 "topic_slug": "how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn",1003 "display_username": "",1004 "primary_group_name": null,1005 "flair_name": null,1006 "flair_url": null,1007 "flair_bg_color": null,1008 "flair_color": null,1009 "flair_group_id": null,1010 "badges_granted": [],1011 "version": 1,1012 "can_edit": false,1013 "can_delete": false,1014 "can_recover": false,1015 "can_see_hidden_post": false,1016 "can_wiki": false,1017 "read": true,1018 "user_title": "",1019 "bookmarked": false,1020 "actions_summary": [],1021 "moderator": true,1022 "admin": true,1023 "staff": true,1024 "user_id": 3534,1025 "hidden": false,1026 "trust_level": 2,1027 "deleted_at": null,1028 "user_deleted": false,1029 "edit_reason": null,1030 "can_view_edit_history": true,1031 "wiki": false,1032 "post_url": "/t/how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn/111980/2",1033 "can_accept_answer": false,1034 "can_unaccept_answer": false,1035 "accepted_answer": false,1036 "topic_accepted_answer": null1037 },1038 {1039 "id": 264229,1040 "name": "Julianolm",1041 "username": "julianolm",1042 "avatar_template": "/user_avatar/discuss.pytorch.org/julianolm/{size}/35135_2.png",1043 "created_at": "2021-02-17T20:25:38.613Z",1044 "cooked": "<p>Thank you for the help!</p>\n<p>Actually, the 1x1 convolutions are in the <code>self.weighted_sum</code> layer in <code>LBCBlock</code>. I’ve tried to follow your advise changing the implementation of this block from</p>\n<pre><code class=\"lang-auto\"> self.conv_filter = nn.Conv2d(n_channels, n_kernels, kernel_size=3, padding=1, bias=False)\n kernels = torch.tensor(new_kernel(n_channels, n_kernels, sparsity)).type('torch.FloatTensor')\n kernels.requires_grad_()\n self.conv_filter.weight = nn.Parameter(kernels)\n self.conv_filter.weight.detach_()\n\n self.weighted_sum = nn.Conv2d(n_kernels, n_channels, kernel_size=1)\n</code></pre>\n<p>to</p>\n<pre><code class=\"lang-auto\"> with torch.no_grad():\n self.conv_filter = nn.Conv2d(n_channels, n_kernels, kernel_size=3, padding=1, bias=False)\n kernels = torch.tensor(new_kernel(n_channels, n_kernels, sparsity)).type('torch.FloatTensor')\n # kernels.requires_grad_()\n self.conv_filter.weight = nn.Parameter(kernels)\n # self.conv_filter.weight.detach_()\n\n self.weighted_sum = nn.Conv2d(n_kernels, n_channels, kernel_size=1)\n</code></pre>\n<p>and also decreasing the batch size from 64 to 16. Nevertheless, the same error still happens, wich I understand as my model is still too large.</p>\n<p>I keep thinking that something is not working the way I expect it to be, since one of the main purpose of using a LBCNN is to have a model with fewer learnable parameters. (If you’re interested: <a href=\"https://arxiv.org/abs/1608.06049\" class=\"inline-onebox\" rel=\"noopener nofollow ugc\">[1608.06049] Local Binary Convolutional Neural Networks</a>)</p>\n<p>Even so, thank you again!</p>",1045 "post_number": 3,1046 "post_type": 1,1047 "posts_count": 4,1048 "updated_at": "2021-02-17T20:25:38.613Z",1049 "reply_count": 1,1050 "reply_to_post_number": 2,1051 "quote_count": 0,1052 "incoming_link_count": 1,1053 "reads": 4,1054 "readers_count": 3,1055 "score": 10.8,1056 "yours": false,1057 "topic_id": 111980,1058 "topic_slug": "how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn",1059 "display_username": "Julianolm",1060 "primary_group_name": null,1061 "flair_name": null,1062 "flair_url": null,1063 "flair_bg_color": null,1064 "flair_color": null,1065 "flair_group_id": null,1066 "badges_granted": [],1067 "version": 1,1068 "can_edit": false,1069 "can_delete": false,1070 "can_recover": false,1071 "can_see_hidden_post": false,1072 "can_wiki": false,1073 "link_counts": [1074 {1075 "url": "https://arxiv.org/abs/1608.06049",1076 "internal": false,1077 "reflection": false,1078 "title": "[1608.06049] Local Binary Convolutional Neural Networks",1079 "clicks": 01080 }1081 ],1082 "read": true,1083 "user_title": null,1084 "reply_to_user": {1085 "id": 3534,1086 "username": "ptrblck",1087 "name": "",1088 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"1089 },1090 "bookmarked": false,1091 "actions_summary": [],1092 "moderator": false,1093 "admin": false,1094 "staff": false,1095 "user_id": 42138,1096 "hidden": false,1097 "trust_level": 1,1098 "deleted_at": null,1099 "user_deleted": false,1100 "edit_reason": null,1101 "can_view_edit_history": true,1102 "wiki": false,1103 "post_url": "/t/how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn/111980/3",1104 "can_accept_answer": false,1105 "can_unaccept_answer": false,1106 "accepted_answer": false,1107 "topic_accepted_answer": null1108 },1109 {1110 "id": 264264,1111 "name": "",1112 "username": "ptrblck",1113 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1114 "created_at": "2021-02-17T23:46:34.244Z",1115 "cooked": "<p>The <code>torch.no_grad()</code> guard should be used in the forward pass and include all operations, which Autograd should not track.<br>\nCould you add it to the <code>forward</code> and check, if this would lower the memory usage?</p>",1116 "post_number": 4,1117 "post_type": 1,1118 "posts_count": 4,1119 "updated_at": "2021-02-17T23:46:34.244Z",1120 "reply_count": 0,1121 "reply_to_post_number": 3,1122 "quote_count": 0,1123 "incoming_link_count": 0,1124 "reads": 4,1125 "readers_count": 3,1126 "score": 0.8,1127 "yours": false,1128 "topic_id": 111980,1129 "topic_slug": "how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn",1130 "display_username": "",1131 "primary_group_name": null,1132 "flair_name": null,1133 "flair_url": null,1134 "flair_bg_color": null,1135 "flair_color": null,1136 "flair_group_id": null,1137 "badges_granted": [],1138 "version": 1,1139 "can_edit": false,1140 "can_delete": false,1141 "can_recover": false,1142 "can_see_hidden_post": false,1143 "can_wiki": false,1144 "read": true,1145 "user_title": "",1146 "reply_to_user": {1147 "id": 42138,1148 "username": "julianolm",1149 "name": "Julianolm",1150 "avatar_template": "/user_avatar/discuss.pytorch.org/julianolm/{size}/35135_2.png"1151 },1152 "bookmarked": false,1153 "actions_summary": [],1154 "moderator": true,1155 "admin": true,1156 "staff": true,1157 "user_id": 3534,1158 "hidden": false,1159 "trust_level": 2,1160 "deleted_at": null,1161 "user_deleted": false,1162 "edit_reason": null,1163 "can_view_edit_history": true,1164 "wiki": false,1165 "post_url": "/t/how-to-resolve-cuda-out-of-memory-on-small-complexity-model-lbcnn/111980/4",1166 "can_accept_answer": false,1167 "can_unaccept_answer": false,1168 "accepted_answer": false,1169 "topic_accepted_answer": null1170 }1171 ],1172 "stream": [1173 263803,1174 263815,1175 264229,1176 2642641177 ]1178 },1179 "timeline_lookup": [1180 [1181 1,1182 17131183 ],1184 [1185 3,1186 17111187 ]1188 ],1189 "suggested_topics": [1190 {1191 "fancy_title": "TypeError: ‘int’ object is not callable for claculating training accuracy",1192 "id": 212609,1193 "title": "TypeError: 'int' object is not callable for claculating training accuracy",1194 "slug": "typeerror-int-object-is-not-callable-for-claculating-training-accuracy",1195 "posts_count": 2,1196 "reply_count": 0,1197 "highest_post_number": 2,1198 "image_url": null,1199 "created_at": "2024-11-06T11:04:54.598Z",1200 "last_posted_at": "2024-11-06T17:20:42.372Z",