Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 459225,7 "name": "Fly Horse24",8 "username": "FlyHorse24",9 "avatar_template": "/user_avatar/discuss.pytorch.org/flyhorse24/{size}/73948_2.png",10 "created_at": "2024-11-12T09:52:56.707Z",11 "cooked": "<p>My program block in “reqs = dist.batch_isend_irecv(ops)\" in p2p.py when send_backward().The send_rank and recv_rank I check are all right.So how can I find the blocking reason?</p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 1,15 "updated_at": "2024-11-12T09:52:56.707Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 4,20 "reads": 7,21 "readers_count": 6,22 "score": 21.4,23 "yours": false,24 "topic_id": 212854,25 "topic_slug": "p2p-blocking-when-pipeline-schedule-communication",26 "display_username": "Fly Horse24",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 1,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "read": true,41 "user_title": null,42 "bookmarked": false,43 "actions_summary": [],44 "moderator": false,45 "admin": false,46 "staff": false,47 "user_id": 80849,48 "hidden": false,49 "trust_level": 0,50 "deleted_at": null,51 "user_deleted": false,52 "edit_reason": null,53 "can_view_edit_history": true,54 "wiki": false,55 "post_url": "/t/p2p-blocking-when-pipeline-schedule-communication/212854/1",56 "can_accept_answer": false,57 "can_unaccept_answer": false,58 "accepted_answer": false,59 "topic_accepted_answer": null,60 "can_vote": false61 }62 ],63 "stream": [64 45922565 ]66 },67 "timeline_lookup": [68 [69 1,70 34771 ]72 ],73 "suggested_topics": [74 {75 "fancy_title": "Too much GPU memory usage for input/model size?",76 "id": 216648,77 "title": "Too much GPU memory usage for input/model size?",78 "slug": "too-much-gpu-memory-usage-for-input-model-size",79 "posts_count": 1,80 "reply_count": 0,81 "highest_post_number": 1,82 "image_url": null,83 "created_at": "2025-02-13T21:04:43.301Z",84 "last_posted_at": "2025-02-13T21:04:43.339Z",85 "bumped": true,86 "bumped_at": "2025-02-14T05:44:02.681Z",87 "archetype": "regular",88 "unseen": false,89 "pinned": false,90 "unpinned": null,91 "visible": true,92 "closed": false,93 "archived": false,94 "bookmarked": null,95 "liked": null,96 "tags_descriptions": {},97 "like_count": 0,98 "views": 38,99 "category_id": 12,100 "featured_link": null,101 "has_accepted_answer": false,102 "posters": [103 {104 "extras": "latest single",105 "description": "Original Poster, Most Recent Poster",106 "user": {107 "id": 82672,108 "username": "reykoki",109 "name": "Rey",110 "avatar_template": "/letter_avatar_proxy/v4/letter/r/a587f6/{size}.png",111 "trust_level": 1112 }113 }114 ]115 },116 {117 "fancy_title": "DTensor sharding strategy support for Autograd override linear op hits issue when bias = None",118 "id": 217129,119 "title": "DTensor sharding strategy support for Autograd override linear op hits issue when bias = None",120 "slug": "dtensor-sharding-strategy-support-for-autograd-override-linear-op-hits-issue-when-bias-none",121 "posts_count": 1,122 "reply_count": 0,123 "highest_post_number": 1,124 "image_url": null,125 "created_at": "2025-02-25T05:14:12.772Z",126 "last_posted_at": "2025-02-25T05:14:12.813Z",127 "bumped": true,128 "bumped_at": "2025-02-25T05:14:12.813Z",129 "archetype": "regular",130 "unseen": false,131 "pinned": false,132 "unpinned": null,133 "visible": true,134 "closed": false,135 "archived": false,136 "bookmarked": null,137 "liked": null,138 "tags_descriptions": {},139 "like_count": 0,140 "views": 87,141 "category_id": 12,142 "featured_link": null,143 "has_accepted_answer": false,144 "posters": [145 {146 "extras": "latest single",147 "description": "Original Poster, Most Recent Poster",148 "user": {149 "id": 82913,150 "username": "satheeshhab",151 "name": "satheesh babu sudarsanan",152 "avatar_template": "/user_avatar/discuss.pytorch.org/satheeshhab/{size}/75859_2.png",153 "trust_level": 0154 }155 }156 ]157 },158 {159 "fancy_title": "How to inference LLM with Multi-GPU",160 "id": 213022,161 "title": "How to inference LLM with Multi-GPU",162 "slug": "how-to-inference-llm-with-multi-gpu",163 "posts_count": 2,164 "reply_count": 0,165 "highest_post_number": 2,166 "image_url": null,167 "created_at": "2024-11-15T19:04:32.869Z",168 "last_posted_at": "2024-11-25T21:58:21.443Z",169 "bumped": true,170 "bumped_at": "2024-11-25T21:58:21.443Z",171 "archetype": "regular",172 "unseen": false,173 "pinned": false,174 "unpinned": null,175 "visible": true,176 "closed": false,177 "archived": false,178 "bookmarked": null,179 "liked": null,180 "tags_descriptions": {},181 "like_count": 0,182 "views": 571,183 "category_id": 12,184 "featured_link": null,185 "has_accepted_answer": false,186 "posters": [187 {188 "extras": null,189 "description": "Original Poster",190 "user": {191 "id": 80831,192 "username": "James_Jang",193 "name": "James Jang",194 "avatar_template": "/user_avatar/discuss.pytorch.org/james_jang/{size}/73925_2.png",195 "trust_level": 1196 }197 },198 {199 "extras": "latest",200 "description": "Most Recent Poster",201 "user": {202 "id": 78505,203 "username": "tianyu",204 "name": "",205 "avatar_template": "/user_avatar/discuss.pytorch.org/tianyu/{size}/72368_2.png",206 "trust_level": 2207 }208 }209 ]210 },211 {212 "fancy_title": "How can I run 5 processes per GPU for three GPUs using DDP?",213 "id": 215534,214 "title": "How can I run 5 processes per GPU for three GPUs using DDP?",215 "slug": "how-can-i-run-5-processes-per-gpu-for-three-gpus-using-ddp",216 "posts_count": 4,217 "reply_count": 2,218 "highest_post_number": 4,219 "image_url": null,220 "created_at": "2025-01-18T02:26:01.681Z",221 "last_posted_at": "2025-01-23T01:38:20.828Z",222 "bumped": true,223 "bumped_at": "2025-01-23T01:38:20.828Z",224 "archetype": "regular",225 "unseen": false,226 "pinned": false,227 "unpinned": null,228 "visible": true,229 "closed": false,230 "archived": false,231 "bookmarked": null,232 "liked": null,233 "tags_descriptions": {},234 "like_count": 2,235 "views": 89,236 "category_id": 12,237 "featured_link": null,238 "has_accepted_answer": true,239 "posters": [240 {241 "extras": "latest",242 "description": "Original Poster, Most Recent Poster",243 "user": {244 "id": 70325,245 "username": "yhl3051",246 "name": "",247 "avatar_template": "/user_avatar/discuss.pytorch.org/yhl3051/{size}/75168_2.png",248 "trust_level": 1249 }250 },251 {252 "extras": null,253 "description": "Frequent Poster, Accepted Answer",254 "user": {255 "id": 3534,256 "username": "ptrblck",257 "name": "",258 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",259 "admin": true,260 "moderator": true,261 "trust_level": 2262 }263 }264 ]265 },266 {267 "fancy_title": "What’s the Best Way to Debug FSDP When Hitting a C++ Backend Error?",268 "id": 219646,269 "title": "What’s the Best Way to Debug FSDP When Hitting a C++ Backend Error?",270 "slug": "what-s-the-best-way-to-debug-fsdp-when-hitting-a-c-backend-error",271 "posts_count": 2,272 "reply_count": 0,273 "highest_post_number": 2,274 "image_url": null,275 "created_at": "2025-05-01T00:51:08.158Z",276 "last_posted_at": "2025-05-01T04:53:36.081Z",277 "bumped": true,278 "bumped_at": "2025-05-01T05:25:41.516Z",279 "archetype": "regular",280 "unseen": false,281 "pinned": false,282 "unpinned": null,283 "visible": true,284 "closed": false,285 "archived": false,286 "bookmarked": null,287 "liked": null,288 "tags_descriptions": {},289 "like_count": 1,290 "views": 82,291 "category_id": 12,292 "featured_link": null,293 "has_accepted_answer": true,294 "posters": [295 {296 "extras": "latest single",297 "description": "Original Poster, Most Recent Poster, Accepted Answer",298 "user": {299 "id": 81938,300 "username": "nariaki3551",301 "name": "Nariaki Tateiwa",302 "avatar_template": "/user_avatar/discuss.pytorch.org/nariaki3551/{size}/74961_2.png",303 "trust_level": 1304 }305 }306 ]307 }308 ],309 "tags_descriptions": {},310 "fancy_title": "P2P blocking when pipeline schedule communication",311 "id": 212854,312 "title": "P2P blocking when pipeline schedule communication",313 "posts_count": 1,314 "created_at": "2024-11-12T09:52:56.645Z",315 "views": 121,316 "reply_count": 0,317 "like_count": 0,318 "last_posted_at": "2024-11-12T09:52:56.707Z",319 "visible": true,320 "closed": false,321 "archived": false,322 "has_summary": false,323 "archetype": "regular",324 "slug": "p2p-blocking-when-pipeline-schedule-communication",325 "category_id": 12,326 "word_count": 30,327 "deleted_at": null,328 "user_id": 80849,329 "featured_link": null,330 "pinned_globally": false,331 "pinned_at": null,332 "pinned_until": null,333 "image_url": null,334 "slow_mode_seconds": 0,335 "draft": null,336 "draft_key": "topic_212854",337 "draft_sequence": null,338 "unpinned": null,339 "pinned": false,340 "current_post_number": 1,341 "highest_post_number": 1,342 "deleted_by": null,343 "actions_summary": [344 {345 "id": 4,346 "count": 0,347 "hidden": false,348 "can_act": false349 },350 {351 "id": 8,352 "count": 0,353 "hidden": false,354 "can_act": false355 },356 {357 "id": 10,358 "count": 0,359 "hidden": false,360 "can_act": false361 },362 {363 "id": 7,364 "count": 0,365 "hidden": false,366 "can_act": false367 }368 ],369 "chunk_size": 20,370 "bookmarked": false,371 "topic_timer": null,372 "message_bus_last_id": 0,373 "participant_count": 1,374 "show_read_indicator": false,375 "thumbnails": null,376 "slow_mode_enabled_until": null,377 "can_vote": false,378 "vote_count": 0,379 "user_voted": false,380 "discourse_zendesk_plugin_zendesk_id": null,381 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",382 "details": {383 "can_edit": false,384 "notification_level": 1,385 "participants": [386 {387 "id": 80849,388 "username": "FlyHorse24",389 "name": "Fly Horse24",390 "avatar_template": "/user_avatar/discuss.pytorch.org/flyhorse24/{size}/73948_2.png",391 "post_count": 1,392 "primary_group_name": null,393 "flair_name": null,394 "flair_url": null,395 "flair_color": null,396 "flair_bg_color": null,397 "flair_group_id": null,398 "trust_level": 0399 }400 ],401 "created_by": {402 "id": 80849,403 "username": "FlyHorse24",404 "name": "Fly Horse24",405 "avatar_template": "/user_avatar/discuss.pytorch.org/flyhorse24/{size}/73948_2.png"406 },407 "last_poster": {408 "id": 80849,409 "username": "FlyHorse24",410 "name": "Fly Horse24",411 "avatar_template": "/user_avatar/discuss.pytorch.org/flyhorse24/{size}/73948_2.png"412 }413 },414 "bookmarks": []415 },416 {417 "post_stream": {418 "posts": [419 {420 "id": 453381,421 "name": "",422 "username": "colibri",423 "avatar_template": "/user_avatar/discuss.pytorch.org/colibri/{size}/72450_2.png",424 "created_at": "2024-08-31T05:50:39.882Z",425 "cooked": "<p>I’ve been experimenting with the new <code>flex_attention</code> module and encountered an issue when trying to integrate it with <code>DistributedDataParallel</code> (DDP). Since <code>flex_attention</code> is a higher-order function, it seems to conflict with DDP’s optimizer.</p>\n<p>Below is a minimal example of my current setup:</p>\n<pre data-code-wrap=\"python\"><code class=\"lang-python\">import os\nimport time\nimport math\n\nimport torch\nfrom torch.nn.parallel import DistributedDataParallel\nfrom torch.nn.attention.flex_attention import flex_attention\n\nclass Model(torch.nn.Module):\n def __init__(self, S, H, D):\n super().__init__()\n\n self.S = S\n self.H = H\n self.D = D\n\n alibi_bias = self.generate_alibi_bias(H)\n self.register_buffer(\"alibi_bias\", alibi_bias, persistent=True)\n self.attention = flex_attention\n\n self.project_qk = torch.nn.Linear(H * D, H * D * 2)\n self.project_v = torch.nn.Linear(H * D, H * D)\n\n def forward(self, hidden_states):\n batch_size, _, _ = hidden_states.size()\n\n query, key = self.project_qk(hidden_states).chunk(2, dim=2)\n query = query.view(self.S, batch_size, self.H, self.D)\n query = query.permute(1, 2, 0, 3)\n\n key = key.view(self.S, batch_size, self.H, self.D)\n key = key.permute(1, 2, 0, 3)\n\n value = self.project_v(hidden_states)\n value = value.view(self.S, batch_size, self.H, self.D)\n value = value.permute(1, 2, 0, 3)\n\n return self.attention(query, key, value, score_mod=self.alibi_score_mod)\n\n def generate_alibi_bias(self, num_heads):\n alibi_bias = [math.exp2(-((i + 1) * 8.0) / num_heads) for i in range(num_heads)]\n return torch.tensor(alibi_bias)\n\n def alibi_score_mod(self, score, b, h, q_idx, kv_idx):\n bias = (q_idx - kv_idx) * self.alibi_bias[h]\n return score + bias\n\nif __name__ == \"__main__\":\n\n B = 64\n H = 12\n S = 512\n D = 64\n\n rank = int(os.environ[\"RANK\"])\n local_rank = int(os.environ[\"LOCAL_RANK\"])\n world_size = int(os.environ[\"WORLD_SIZE\"])\n\n torch.distributed.init_process_group(backend=\"nccl\", rank=rank, world_size=world_size)\n\n torch.cuda.set_device(local_rank)\n device = torch.device(\"cuda\", local_rank)\n\n model = Model(S, H, D)\n model.to(device)\n model = DistributedDataParallel(model, device_ids=[local_rank])\n torch.compile(model)\n\n for i in range(100):\n start = time.perf_counter()\n hidden_states = torch.randn(B, S, H * D).to(device)\n attention_scores = model(hidden_states)\n torch.cuda.synchronize()\n print(f\"{i}: {time.perf_counter() - start:.4f}\")\n</code></pre>\n<p>I run the script using the following command:</p>\n<pre data-code-wrap=\"bash\"><code class=\"lang-bash\">torchrun --standalone --nnodes=1 --nproc_per_node=1 flex_attention_test.py\n</code></pre>\n<p>However, I encounter the following error:</p>\n<pre><code class=\"lang-auto\">[rank0]: File \"/home/colibri/mambaforge/envs/pytorch2_5/lib/python3.11/site-packages/torch/_dynamo/output_graph.py\", line 1457, in _call_user_compiler\n[rank0]: raise BackendCompilerFailed(self.compiler_fn, e) from e\n[rank0]: torch._dynamo.exc.BackendCompilerFailed: backend='compile_fn' raised:\n[rank0]: NotImplementedError: DDPOptimizer backend: Found a higher order op in the graph. This is not supported. Please turn off DDP optimizer using torch._dynamo.config.optimize_ddp=False. Note that this can cause performance degradation because there will be one bucket for the entire Dynamo graph. Please refer to this issue - https://github.com/pytorch/pytorch/issues/104674.\n</code></pre>\n<p>Disabling the DDP optimizer resolves the error but results in significant performance degradation.</p>\n<p>I’m seeking guidance on whether there’s a proper way to use <code>flex_attention</code> or similar higher-order operations in conjunction with DDP without sacrificing performance. Any advice or insights would be greatly appreciated.</p>",426 "post_number": 1,427 "post_type": 1,428 "posts_count": 3,429 "updated_at": "2024-08-31T05:50:39.882Z",430 "reply_count": 0,431 "reply_to_post_number": null,432 "quote_count": 0,433 "incoming_link_count": 597,434 "reads": 19,435 "readers_count": 18,436 "score": 3028.8,437 "yours": false,438 "topic_id": 208916,439 "topic_slug": "using-flex-attention-with-distributeddataparallel-ddp",440 "display_username": "",441 "primary_group_name": null,442 "flair_name": null,443 "flair_url": null,444 "flair_bg_color": null,445 "flair_color": null,446 "flair_group_id": null,447 "badges_granted": [],448 "version": 1,449 "can_edit": false,450 "can_delete": false,451 "can_recover": false,452 "can_see_hidden_post": false,453 "can_wiki": false,454 "read": true,455 "user_title": null,456 "bookmarked": false,457 "actions_summary": [458 {459 "id": 2,460 "count": 4461 }462 ],463 "moderator": false,464 "admin": false,465 "staff": false,466 "user_id": 78603,467 "hidden": false,468 "trust_level": 0,469 "deleted_at": null,470 "user_deleted": false,471 "edit_reason": null,472 "can_view_edit_history": true,473 "wiki": false,474 "post_url": "/t/using-flex-attention-with-distributeddataparallel-ddp/208916/1",475 "can_accept_answer": false,476 "can_unaccept_answer": false,477 "accepted_answer": false,478 "topic_accepted_answer": null,479 "can_vote": false480 },481 {482 "id": 453719,483 "name": "Sam Galanakis",484 "username": "Sam_Galanakis",485 "avatar_template": "/user_avatar/discuss.pytorch.org/sam_galanakis/{size}/30002_2.png",486 "created_at": "2024-09-05T10:13:43.277Z",487 "cooked": "<p>Also having exact same issue could only ge tit working with disabled ddp optimization.</p>",488 "post_number": 2,489 "post_type": 1,490 "posts_count": 3,491 "updated_at": "2024-09-05T10:13:43.277Z",492 "reply_count": 1,493 "reply_to_post_number": null,494 "quote_count": 0,495 "incoming_link_count": 3,496 "reads": 16,497 "readers_count": 15,498 "score": 53.2,499 "yours": false,500 "topic_id": 208916,501 "topic_slug": "using-flex-attention-with-distributeddataparallel-ddp",502 "display_username": "Sam Galanakis",503 "primary_group_name": null,504 "flair_name": null,505 "flair_url": null,506 "flair_bg_color": null,507 "flair_color": null,508 "flair_group_id": null,509 "badges_granted": [],510 "version": 1,511 "can_edit": false,512 "can_delete": false,513 "can_recover": false,514 "can_see_hidden_post": false,515 "can_wiki": false,516 "read": true,517 "user_title": null,518 "bookmarked": false,519 "actions_summary": [520 {521 "id": 2,522 "count": 2523 }524 ],525 "moderator": false,526 "admin": false,527 "staff": false,528 "user_id": 42137,529 "hidden": false,530 "trust_level": 1,531 "deleted_at": null,532 "user_deleted": false,533 "edit_reason": null,534 "can_view_edit_history": true,535 "wiki": false,536 "post_url": "/t/using-flex-attention-with-distributeddataparallel-ddp/208916/2",537 "can_accept_answer": false,538 "can_unaccept_answer": false,539 "accepted_answer": false,540 "topic_accepted_answer": null541 },542 {543 "id": 459218,544 "name": "",545 "username": "zbh2047",546 "avatar_template": "/letter_avatar_proxy/v4/letter/z/b5ac83/{size}.png",547 "created_at": "2024-11-12T08:16:42.797Z",548 "cooked": "<p>Also having exactly the same issue</p>",549 "post_number": 3,550 "post_type": 1,551 "posts_count": 3,552 "updated_at": "2024-11-12T08:16:42.797Z",553 "reply_count": 0,554 "reply_to_post_number": 2,555 "quote_count": 0,556 "incoming_link_count": 5,557 "reads": 11,558 "readers_count": 10,559 "score": 27.2,560 "yours": false,561 "topic_id": 208916,562 "topic_slug": "using-flex-attention-with-distributeddataparallel-ddp",563 "display_username": "",564 "primary_group_name": null,565 "flair_name": null,566 "flair_url": null,567 "flair_bg_color": null,568 "flair_color": null,569 "flair_group_id": null,570 "badges_granted": [],571 "version": 1,572 "can_edit": false,573 "can_delete": false,574 "can_recover": false,575 "can_see_hidden_post": false,576 "can_wiki": false,577 "read": true,578 "user_title": null,579 "reply_to_user": {580 "id": 42137,581 "username": "Sam_Galanakis",582 "name": "Sam Galanakis",583 "avatar_template": "/user_avatar/discuss.pytorch.org/sam_galanakis/{size}/30002_2.png"584 },585 "bookmarked": false,586 "actions_summary": [],587 "moderator": false,588 "admin": false,589 "staff": false,590 "user_id": 19694,591 "hidden": false,592 "trust_level": 1,593 "deleted_at": null,594 "user_deleted": false,595 "edit_reason": null,596 "can_view_edit_history": true,597 "wiki": false,598 "post_url": "/t/using-flex-attention-with-distributeddataparallel-ddp/208916/3",599 "can_accept_answer": false,600 "can_unaccept_answer": false,601 "accepted_answer": false,602 "topic_accepted_answer": null603 }604 ],605 "stream": [606 453381,607 453719,608 459218609 ]610 },611 "timeline_lookup": [612 [613 1,614 421615 ],616 [617 2,618 415619 ],620 [621 3,622 347623 ]624 ],625 "suggested_topics": [626 {627 "fancy_title": "Maybe there is a precision issue with torch.quantize_per_channel?",628 "id": 212389,629 "title": "Maybe there is a precision issue with torch.quantize_per_channel?",630 "slug": "maybe-there-is-a-precision-issue-with-torch-quantize-per-channel",631 "posts_count": 4,632 "reply_count": 2,633 "highest_post_number": 5,634 "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/8/0/80f2c0d04c4f656428959b361ec665e34314eb4f.png",635 "created_at": "2024-11-01T03:11:47.675Z",636 "last_posted_at": "2024-11-08T02:31:12.293Z",637 "bumped": true,638 "bumped_at": "2024-11-08T02:31:12.293Z",639 "archetype": "regular",640 "unseen": false,641 "pinned": false,642 "unpinned": null,643 "visible": true,644 "closed": false,645 "archived": false,646 "bookmarked": null,647 "liked": null,648 "tags_descriptions": {},649 "like_count": 1,650 "views": 48,651 "category_id": 1,652 "featured_link": null,653 "has_accepted_answer": false,654 "posters": [655 {656 "extras": null,657 "description": "Original Poster",658 "user": {659 "id": 80599,660 "username": "zfac",661 "name": "Zfac",662 "avatar_template": "/user_avatar/discuss.pytorch.org/zfac/{size}/73683_2.png",663 "trust_level": 1664 }665 },666 {667 "extras": "latest",668 "description": "Most Recent Poster",669 "user": {670 "id": 56768,671 "username": "yushu.gao",672 "name": "",673 "avatar_template": "/user_avatar/discuss.pytorch.org/yushu.gao/{size}/50534_2.png",674 "trust_level": 1675 }676 }677 ]678 },679 {680 "fancy_title": "Use `torch.Tensor` as typehint",681 "id": 219334,682 "title": "Use `torch.Tensor` as typehint",683 "slug": "use-torch-tensor-as-typehint",684 "posts_count": 1,685 "reply_count": 0,686 "highest_post_number": 1,687 "image_url": null,688 "created_at": "2025-04-22T10:42:08.655Z",689 "last_posted_at": "2025-04-22T10:42:08.698Z",690 "bumped": true,691 "bumped_at": "2025-04-22T10:42:08.698Z",692 "archetype": "regular",693 "unseen": false,694 "pinned": false,695 "unpinned": null,696 "visible": true,697 "closed": false,698 "archived": false,699 "bookmarked": null,700 "liked": null,701 "tags_descriptions": {},702 "like_count": 0,703 "views": 198,704 "category_id": 1,705 "featured_link": null,706 "has_accepted_answer": false,707 "posters": [708 {709 "extras": "latest single",710 "description": "Original Poster, Most Recent Poster",711 "user": {712 "id": 45130,713 "username": "pascalm",714 "name": "",715 "avatar_template": "/user_avatar/discuss.pytorch.org/pascalm/{size}/76741_2.png",716 "trust_level": 1717 }718 }719 ]720 },721 {722 "fancy_title": "Suggestion for ML approach to mimic FEM results?",723 "id": 213993,724 "title": "Suggestion for ML approach to mimic FEM results?",725 "slug": "suggestion-for-ml-approach-to-mimic-fem-results",726 "posts_count": 1,727 "reply_count": 0,728 "highest_post_number": 1,729 "image_url": null,730 "created_at": "2024-12-09T12:34:40.329Z",731 "last_posted_at": "2024-12-09T12:34:40.382Z",732 "bumped": true,733 "bumped_at": "2024-12-09T13:19:10.610Z",734 "archetype": "regular",735 "unseen": false,736 "pinned": false,737 "unpinned": null,738 "visible": true,739 "closed": false,740 "archived": false,741 "bookmarked": null,742 "liked": null,743 "tags_descriptions": {},744 "like_count": 0,745 "views": 46,746 "category_id": 1,747 "featured_link": null,748 "has_accepted_answer": false,749 "posters": [750 {751 "extras": "latest single",752 "description": "Original Poster, Most Recent Poster",753 "user": {754 "id": 81403,755 "username": "LainuUrdin",756 "name": "Laino Urdin",757 "avatar_template": "/user_avatar/discuss.pytorch.org/lainuurdin/{size}/74437_2.png",758 "trust_level": 0759 }760 }761 ]762 },763 {764 "fancy_title": "Flex attention and SDPA output natively equivalent?",765 "id": 214746,766 "title": "Flex attention and SDPA output natively equivalent?",767 "slug": "flex-attention-and-sdpa-output-natively-equivalent",768 "posts_count": 3,769 "reply_count": 1,770 "highest_post_number": 3,771 "image_url": null,772 "created_at": "2024-12-29T06:02:12.151Z",773 "last_posted_at": "2024-12-30T03:50:37.382Z",774 "bumped": true,775 "bumped_at": "2024-12-30T03:50:37.382Z",776 "archetype": "regular",777 "unseen": false,778 "pinned": false,779 "unpinned": null,780 "visible": true,781 "closed": false,782 "archived": false,783 "bookmarked": null,784 "liked": null,785 "tags_descriptions": {},786 "like_count": 0,787 "views": 686,788 "category_id": 1,789 "featured_link": null,790 "has_accepted_answer": true,791 "posters": [792 {793 "extras": "latest",794 "description": "Original Poster, Most Recent Poster",795 "user": {796 "id": 81765,797 "username": "lhallee",798 "name": "Logan Hallee",799 "avatar_template": "/user_avatar/discuss.pytorch.org/lhallee/{size}/74777_2.png",800 "trust_level": 0801 }802 },803 {804 "extras": null,805 "description": "Frequent Poster, Accepted Answer",806 "user": {807 "id": 41396,808 "username": "soulitzer",809 "name": "",810 "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",811 "trust_level": 2812 }813 }814 ]815 },816 {817 "fancy_title": "KAN - Model not replicated to all GPUs with nn.DataParallel()",818 "id": 220084,819 "title": "KAN - Model not replicated to all GPUs with nn.DataParallel()",820 "slug": "kan-model-not-replicated-to-all-gpus-with-nn-dataparallel",821 "posts_count": 5,822 "reply_count": 3,823 "highest_post_number": 5,824 "image_url": null,825 "created_at": "2025-05-15T21:21:29.631Z",826 "last_posted_at": "2025-05-17T03:32:33.555Z",827 "bumped": true,828 "bumped_at": "2025-05-17T03:32:33.555Z",829 "archetype": "regular",830 "unseen": false,831 "pinned": false,832 "unpinned": null,833 "visible": true,834 "closed": false,835 "archived": false,836 "bookmarked": null,837 "liked": null,838 "tags_descriptions": {},839 "like_count": 0,840 "views": 155,841 "category_id": 1,842 "featured_link": null,843 "has_accepted_answer": false,844 "posters": [845 {846 "extras": "latest",847 "description": "Original Poster, Most Recent Poster",848 "user": {849 "id": 84306,850 "username": "Aditya_Ratan",851 "name": "Aditya Ratan",852 "avatar_template": "/user_avatar/discuss.pytorch.org/aditya_ratan/{size}/74867_2.png",853 "trust_level": 0854 }855 },856 {857 "extras": null,858 "description": "Frequent Poster",859 "user": {860 "id": 3534,861 "username": "ptrblck",862 "name": "",863 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",864 "admin": true,865 "moderator": true,866 "trust_level": 2867 }868 }869 ]870 }871 ],872 "tags_descriptions": {},873 "fancy_title": "Using `flex_attention` with DistributedDataParallel (DDP)",874 "id": 208916,875 "title": "Using `flex_attention` with DistributedDataParallel (DDP)",876 "posts_count": 3,877 "created_at": "2024-08-31T05:50:39.822Z",878 "views": 718,879 "reply_count": 1,880 "like_count": 6,881 "last_posted_at": "2024-11-12T08:16:42.797Z",882 "visible": true,883 "closed": false,884 "archived": false,885 "has_summary": false,886 "archetype": "regular",887 "slug": "using-flex-attention-with-distributeddataparallel-ddp",888 "category_id": 1,889 "word_count": 514,890 "deleted_at": null,891 "user_id": 78603,892 "featured_link": null,893 "pinned_globally": false,894 "pinned_at": null,895 "pinned_until": null,896 "image_url": null,897 "slow_mode_seconds": 0,898 "draft": null,899 "draft_key": "topic_208916",900 "draft_sequence": null,901 "unpinned": null,902 "pinned": false,903 "current_post_number": 1,904 "highest_post_number": 3,905 "deleted_by": null,906 "actions_summary": [907 {908 "id": 4,909 "count": 0,910 "hidden": false,911 "can_act": false912 },913 {914 "id": 8,915 "count": 0,916 "hidden": false,917 "can_act": false918 },919 {920 "id": 10,921 "count": 0,922 "hidden": false,923 "can_act": false924 },925 {926 "id": 7,927 "count": 0,928 "hidden": false,929 "can_act": false930 }931 ],932 "chunk_size": 20,933 "bookmarked": false,934 "topic_timer": null,935 "message_bus_last_id": 0,936 "participant_count": 3,937 "show_read_indicator": false,938 "thumbnails": null,939 "slow_mode_enabled_until": null,940 "can_vote": false,941 "vote_count": 0,942 "user_voted": false,943 "discourse_zendesk_plugin_zendesk_id": null,944 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",945 "details": {946 "can_edit": false,947 "notification_level": 1,948 "participants": [949 {950 "id": 19694,951 "username": "zbh2047",952 "name": "",953 "avatar_template": "/letter_avatar_proxy/v4/letter/z/b5ac83/{size}.png",954 "post_count": 1,955 "primary_group_name": null,956 "flair_name": null,957 "flair_url": null,958 "flair_color": null,959 "flair_bg_color": null,960 "flair_group_id": null,961 "trust_level": 1962 },963 {964 "id": 42137,965 "username": "Sam_Galanakis",966 "name": "Sam Galanakis",967 "avatar_template": "/user_avatar/discuss.pytorch.org/sam_galanakis/{size}/30002_2.png",968 "post_count": 1,969 "primary_group_name": null,970 "flair_name": null,971 "flair_url": null,972 "flair_color": null,973 "flair_bg_color": null,974 "flair_group_id": null,975 "trust_level": 1976 },977 {978 "id": 78603,979 "username": "colibri",980 "name": "",981 "avatar_template": "/user_avatar/discuss.pytorch.org/colibri/{size}/72450_2.png",982 "post_count": 1,983 "primary_group_name": null,984 "flair_name": null,985 "flair_url": null,986 "flair_color": null,987 "flair_bg_color": null,988 "flair_group_id": null,989 "trust_level": 0990 }991 ],992 "created_by": {993 "id": 78603,994 "username": "colibri",995 "name": "",996 "avatar_template": "/user_avatar/discuss.pytorch.org/colibri/{size}/72450_2.png"997 },998 "last_poster": {999 "id": 19694,1000 "username": "zbh2047",1001 "name": "",1002 "avatar_template": "/letter_avatar_proxy/v4/letter/z/b5ac83/{size}.png"1003 }1004 },1005 "bookmarks": []1006 },1007 {1008 "post_stream": {1009 "posts": [1010 {1011 "id": 459155,1012 "name": "Tony Robinson",1013 "username": "tonyr",1014 "avatar_template": "/user_avatar/discuss.pytorch.org/tonyr/{size}/74253_2.png",1015 "created_at": "2024-11-11T15:10:47.382Z",1016 "cooked": "<p>The basic view of autograd is that Parameters are held as leaf variables and can be combined in lots of wonderful and weird ways. We call backward() on a result and the gradients are computed all the way back to every leaf variable.</p>\n<p>However, let’s say that I want to use a function of some parameters, f(\\theta) in each pass of my minibatch. f() is expensive, and f(\\theta) won’t change until I step the optimiser. So I’d like to think that I could say y = f(\\theta) after zero_grad and before all passes and then use y within the minibatch loop. The idea is to accumulate gradients on y in the minibatch and not on the leaf variable, \\theta.</p>\n<p>Is autograd smart enough to see that y is constant in the minibatch loop, or do I need to have two optimisers, one over what changes in the minibatch and one to optimise f(\\theta).</p>\n<p>Where do I look to learn more?</p>\n<p>Example code below just to illustrate what I’m talking about</p>\n<pre><code class=\"lang-plaintext\">import torch\n\nninp = 8\nnout = 16\nnmini = 8\ntheta =\ttorch.nn.Parameter(torch.zeros(ninp))\nf = torch.nn.Linear(ninp, nout)\n\nwhile True:\n\n # zero_grad() \n\n y = f(theta)\n vec = torch.ones(nout)\n for _ in range(nmini):\n vec = y * vec\n\n loss = vec.sum()\n loss.backward()\n\n # step optimiser \n</code></pre>",1017 "post_number": 1,1018 "post_type": 1,1019 "posts_count": 3,1020 "updated_at": "2024-11-12T07:23:17.594Z",1021 "reply_count": 1,1022 "reply_to_post_number": null,1023 "quote_count": 0,1024 "incoming_link_count": 136,1025 "reads": 9,1026 "readers_count": 8,1027 "score": 686.8,1028 "yours": false,1029 "topic_id": 212808,1030 "topic_slug": "minibatch-and-efficient-gradient-accumulation",1031 "display_username": "Tony Robinson",1032 "primary_group_name": null,1033 "flair_name": null,1034 "flair_url": null,1035 "flair_bg_color": null,1036 "flair_color": null,1037 "flair_group_id": null,1038 "badges_granted": [],1039 "version": 4,1040 "can_edit": false,1041 "can_delete": false,1042 "can_recover": false,1043 "can_see_hidden_post": false,1044 "can_wiki": false,1045 "read": true,1046 "user_title": null,1047 "bookmarked": false,1048 "actions_summary": [],1049 "moderator": false,1050 "admin": false,1051 "staff": false,1052 "user_id": 30062,1053 "hidden": false,1054 "trust_level": 2,1055 "deleted_at": null,1056 "user_deleted": false,1057 "edit_reason": null,1058 "can_view_edit_history": true,1059 "wiki": false,1060 "post_url": "/t/minibatch-and-efficient-gradient-accumulation/212808/1",1061 "can_accept_answer": false,1062 "can_unaccept_answer": false,1063 "accepted_answer": false,1064 "topic_accepted_answer": null,1065 "can_vote": false1066 },1067 {1068 "id": 459175,1069 "name": "",1070 "username": "ptrblck",1071 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",1072 "created_at": "2024-11-11T22:33:51.935Z",1073 "cooked": "<aside class=\"quote no-group\" data-username=\"tonyr\" data-post=\"1\" data-topic=\"212808\">\n<div class=\"title\">\n<div class=\"quote-controls\"></div>\n<img loading=\"lazy\" alt=\"\" width=\"24\" height=\"24\" src=\"https://discuss.pytorch.org/user_avatar/discuss.pytorch.org/tonyr/48/74253_2.png\" class=\"avatar\"> tonyr:</div>\n<blockquote>\n<p>Is autograd smart enough to see that y is constant in the minibatch loop</p>\n</blockquote>\n</aside>\n<p><code>y</code> is not constant as you are re-assigning the result of <code>y * vec</code> to it while the multiplication is differentiable. Autograd will thus create a computation graph with these multiplications.</p>",1074 "post_number": 2,1075 "post_type": 1,1076 "posts_count": 3,1077 "updated_at": "2024-11-11T22:33:51.935Z",1078 "reply_count": 1,1079 "reply_to_post_number": null,1080 "quote_count": 1,1081 "incoming_link_count": 1,1082 "reads": 7,1083 "readers_count": 6,1084 "score": 11.4,1085 "yours": false,1086 "topic_id": 212808,1087 "topic_slug": "minibatch-and-efficient-gradient-accumulation",1088 "display_username": "",1089 "primary_group_name": null,1090 "flair_name": null,1091 "flair_url": null,1092 "flair_bg_color": null,1093 "flair_color": null,1094 "flair_group_id": null,1095 "badges_granted": [],1096 "version": 1,1097 "can_edit": false,1098 "can_delete": false,1099 "can_recover": false,1100 "can_see_hidden_post": false,1101 "can_wiki": false,1102 "read": true,1103 "user_title": "",1104 "bookmarked": false,1105 "actions_summary": [],1106 "moderator": true,1107 "admin": true,1108 "staff": true,1109 "user_id": 3534,1110 "hidden": false,1111 "trust_level": 2,1112 "deleted_at": null,1113 "user_deleted": false,1114 "edit_reason": null,1115 "can_view_edit_history": true,1116 "wiki": false,1117 "post_url": "/t/minibatch-and-efficient-gradient-accumulation/212808/2",1118 "can_accept_answer": false,1119 "can_unaccept_answer": false,1120 "accepted_answer": false,1121 "topic_accepted_answer": null1122 },1123 {1124 "id": 459215,1125 "name": "Tony Robinson",1126 "username": "tonyr",1127 "avatar_template": "/user_avatar/discuss.pytorch.org/tonyr/{size}/74253_2.png",1128 "created_at": "2024-11-12T07:18:43.218Z",1129 "cooked": "<p>Thanks <a class=\"mention\" href=\"/u/ptrblck\">@ptrblck</a>, you are right, I’d screwed up my example that was supposed to illustrate the problem. I’ve now edited the code to swap vec and y in the loop, so y is a constant in the loop. I’ve also edited it to start with a basic view of autograd as an introduction to the problem.</p>\n<p>The basic view is as you say, y is a function of my parameters and so isn’t a constant. However, the question is specifically within a mini-batch setting, that is I’m computing y outside of the minibatch then iterating over my inputs/computation within the minibatch. Within this setting, y is a constant within the mini-batch loop and there is a gain to be made in computations by accumulating the gradients at y and backproping to \\theta only once.</p>\n<p>I’ve only used the mini-batch scenario for clarity as it shows that backward() is called many times but there is a point in the computation graph where gradients could be accumulated and one call of backward() made from there on. Maybe it would have been clearer if I had said:</p>\n<pre><code class=\"lang-plaintext\"> y = f(theta)\n g(y).backward()\n h(y).backward()\n</code></pre>\n<p>where I’d like to call f(theta).backward() only once. Or there again this might have ended up in another (related) discussion about backproping through the graph twice…</p>\n<p>Of course I’m working on it as I type. Now I’ve got this far, I’m pretty sure that autograd can not make this optimisation. It can only fire when backward() is called, so there is no ‘end of loop’ call for it to complete on. So what I’m looking for is a way to introduce a manual graph break so that y looks like a Parameter within the loop and then make one more call using those gradients. Something like:</p>\n<pre><code class=\"lang-plaintext\"> y = Parameter(f(theta).detach())\n for _ in range(...)\n # use y\n loss.backward(retain_graph=True)\n f.weight.grad += y.grad\n \n # some optimiser call that uses f.weight.grad\n</code></pre>\n<p>This seems like something we’d like to do quite a lot in big models, I’m just missing the right search terms to find the answer.</p>",1130 "post_number": 3,1131 "post_type": 1,1132 "posts_count": 3,1133 "updated_at": "2024-11-12T07:27:50.619Z",1134 "reply_count": 0,1135 "reply_to_post_number": 2,1136 "quote_count": 0,1137 "incoming_link_count": 2,1138 "reads": 6,1139 "readers_count": 5,1140 "score": 11.2,1141 "yours": false,1142 "topic_id": 212808,1143 "topic_slug": "minibatch-and-efficient-gradient-accumulation",1144 "display_username": "Tony Robinson",1145 "primary_group_name": null,1146 "flair_name": null,1147 "flair_url": null,1148 "flair_bg_color": null,1149 "flair_color": null,1150 "flair_group_id": null,1151 "badges_granted": [],1152 "version": 3,1153 "can_edit": false,1154 "can_delete": false,1155 "can_recover": false,1156 "can_see_hidden_post": false,1157 "can_wiki": false,1158 "read": true,1159 "user_title": null,1160 "reply_to_user": {1161 "id": 3534,1162 "username": "ptrblck",1163 "name": "",1164 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png"1165 },1166 "bookmarked": false,1167 "actions_summary": [],1168 "moderator": false,1169 "admin": false,1170 "staff": false,1171 "user_id": 30062,1172 "hidden": false,1173 "trust_level": 2,1174 "deleted_at": null,1175 "user_deleted": false,1176 "edit_reason": null,1177 "can_view_edit_history": true,1178 "wiki": false,1179 "post_url": "/t/minibatch-and-efficient-gradient-accumulation/212808/3",1180 "can_accept_answer": false,1181 "can_unaccept_answer": false,1182 "accepted_answer": false,1183 "topic_accepted_answer": null1184 }1185 ],1186 "stream": [1187 459155,1188 459175,1189 4592151190 ]1191 },1192 "timeline_lookup": [1193 [1194 1,1195 3481196 ],1197 [1198 3,1199 3471200 ]