Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 403389,7 "name": "Yihua Xu",8 "username": "Yihua_Xu",9 "avatar_template": "/user_avatar/discuss.pytorch.org/yihua_xu/{size}/59111_2.png",10 "created_at": "2023-05-25T04:45:01.836Z",11 "cooked": "<p>I am a freshman on using DDP, so I am trying to run an example supported by Pytorch. Its link is <a href=\"https://github.com/pytorch/examples/blob/main/distributed/ddp-tutorial-series/multigpu.py\" rel=\"noopener nofollow ugc\">multigpu.py</a>. There are 4 GPUs on my machine. PyTorch version: ‘2.0.1+cu117’; OS: Ubuntu 20.04.4 LTS<br>\nWhen I run the code with the following command format <code>python multigpu.py 10 5 > output.tx &</code>, it will spend several hours without output, even without errors. I know it did run because it created a new file named “checkpoint.pt” It should finish in a minute and print out something like <code>[GPU 0 ...</code> <code>[GPU 1 ...</code> <code>[GPU 2 ...</code> <code>[GPU 3 ...</code><br>\nCould someone help me?</p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 1,15 "updated_at": "2023-05-25T04:45:01.836Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 21,20 "reads": 4,21 "readers_count": 3,22 "score": 105.8,23 "yours": false,24 "topic_id": 180658,25 "topic_slug": "no-ouput-by-running-the-pytorch-ddp-example",26 "display_username": "Yihua Xu",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 1,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "link_counts": [41 {42 "url": "https://github.com/pytorch/examples/blob/main/distributed/ddp-tutorial-series/multigpu.py",43 "internal": false,44 "reflection": false,45 "title": "examples/multigpu.py at main · pytorch/examples · GitHub",46 "clicks": 147 }48 ],49 "read": true,50 "user_title": null,51 "bookmarked": false,52 "actions_summary": [],53 "moderator": false,54 "admin": false,55 "staff": false,56 "user_id": 66450,57 "hidden": false,58 "trust_level": 1,59 "deleted_at": null,60 "user_deleted": false,61 "edit_reason": null,62 "can_view_edit_history": true,63 "wiki": false,64 "post_url": "/t/no-ouput-by-running-the-pytorch-ddp-example/180658/1",65 "can_accept_answer": false,66 "can_unaccept_answer": false,67 "accepted_answer": false,68 "topic_accepted_answer": null,69 "can_vote": false70 }71 ],72 "stream": [73 40338974 ]75 },76 "timeline_lookup": [77 [78 1,79 88580 ]81 ],82 "suggested_topics": [83 {84 "fancy_title": "Output different with and without FSDP",85 "id": 213302,86 "title": "Output different with and without FSDP",87 "slug": "output-different-with-and-without-fsdp",88 "posts_count": 2,89 "reply_count": 0,90 "highest_post_number": 2,91 "image_url": null,92 "created_at": "2024-11-22T11:24:28.244Z",93 "last_posted_at": "2024-11-25T22:08:33.407Z",94 "bumped": true,95 "bumped_at": "2024-11-25T22:08:33.407Z",96 "archetype": "regular",97 "unseen": false,98 "pinned": false,99 "unpinned": null,100 "visible": true,101 "closed": false,102 "archived": false,103 "bookmarked": null,104 "liked": null,105 "tags_descriptions": {},106 "like_count": 0,107 "views": 35,108 "category_id": 12,109 "featured_link": null,110 "has_accepted_answer": false,111 "posters": [112 {113 "extras": null,114 "description": "Original Poster",115 "user": {116 "id": 40210,117 "username": "Abhiraj_Kanse",118 "name": "Abhiraj Kanse",119 "avatar_template": "/user_avatar/discuss.pytorch.org/abhiraj_kanse/{size}/32483_2.png",120 "trust_level": 1121 }122 },123 {124 "extras": "latest",125 "description": "Most Recent Poster",126 "user": {127 "id": 78505,128 "username": "tianyu",129 "name": "",130 "avatar_template": "/user_avatar/discuss.pytorch.org/tianyu/{size}/72368_2.png",131 "trust_level": 2132 }133 }134 ]135 },136 {137 "fancy_title": "How to solve the issue with getting free ports in Pytorch DDP?",138 "id": 213921,139 "title": "How to solve the issue with getting free ports in Pytorch DDP?",140 "slug": "how-to-solve-the-issue-with-getting-free-ports-in-pytorch-ddp",141 "posts_count": 1,142 "reply_count": 0,143 "highest_post_number": 1,144 "image_url": null,145 "created_at": "2024-12-06T19:39:09.953Z",146 "last_posted_at": "2024-12-06T19:39:09.998Z",147 "bumped": true,148 "bumped_at": "2024-12-06T19:39:09.998Z",149 "archetype": "regular",150 "unseen": false,151 "pinned": false,152 "unpinned": null,153 "visible": true,154 "closed": false,155 "archived": false,156 "bookmarked": null,157 "liked": null,158 "tags_descriptions": {},159 "like_count": 0,160 "views": 100,161 "category_id": 12,162 "featured_link": null,163 "has_accepted_answer": false,164 "posters": [165 {166 "extras": "latest single",167 "description": "Original Poster, Most Recent Poster",168 "user": {169 "id": 81360,170 "username": "Shataneek",171 "name": "",172 "avatar_template": "/letter_avatar_proxy/v4/letter/s/57b2e6/{size}.png",173 "trust_level": 0174 }175 }176 ]177 },178 {179 "fancy_title": "Issue with Training Loop Using DDP and AMP: Process Getting Stuck",180 "id": 214753,181 "title": "Issue with Training Loop Using DDP and AMP: Process Getting Stuck",182 "slug": "issue-with-training-loop-using-ddp-and-amp-process-getting-stuck",183 "posts_count": 5,184 "reply_count": 2,185 "highest_post_number": 5,186 "image_url": null,187 "created_at": "2024-12-29T15:47:51.738Z",188 "last_posted_at": "2024-12-30T10:30:21.851Z",189 "bumped": true,190 "bumped_at": "2024-12-30T10:30:21.851Z",191 "archetype": "regular",192 "unseen": false,193 "pinned": false,194 "unpinned": null,195 "visible": true,196 "closed": false,197 "archived": false,198 "bookmarked": null,199 "liked": null,200 "tags_descriptions": {},201 "like_count": 0,202 "views": 161,203 "category_id": 12,204 "featured_link": null,205 "has_accepted_answer": false,206 "posters": [207 {208 "extras": "latest",209 "description": "Original Poster, Most Recent Poster",210 "user": {211 "id": 69415,212 "username": "Scbas_scias",213 "name": "Scbas scias",214 "avatar_template": "/user_avatar/discuss.pytorch.org/scbas_scias/{size}/61637_2.png",215 "trust_level": 1216 }217 },218 {219 "extras": null,220 "description": "Frequent Poster",221 "user": {222 "id": 41396,223 "username": "soulitzer",224 "name": "",225 "avatar_template": "/letter_avatar_proxy/v4/letter/s/839c29/{size}.png",226 "trust_level": 2227 }228 },229 {230 "extras": null,231 "description": "Frequent Poster",232 "user": {233 "id": 81709,234 "username": "QLYYLQ",235 "name": "qly",236 "avatar_template": "/user_avatar/discuss.pytorch.org/qlyylq/{size}/74718_2.png",237 "trust_level": 1238 }239 }240 ]241 },242 {243 "fancy_title": "Torch.distributed.all_reduce causes memory trashing",244 "id": 215024,245 "title": "Torch.distributed.all_reduce causes memory trashing",246 "slug": "torch-distributed-all-reduce-causes-memory-trashing",247 "posts_count": 3,248 "reply_count": 1,249 "highest_post_number": 3,250 "image_url": "https://discuss.pytorch.org/uploads/default/optimized/3X/5/8/5838d755d5ca5296ddc8137c78a5014be93befef_2_1024x422.jpeg",251 "created_at": "2025-01-06T12:06:07.456Z",252 "last_posted_at": "2025-01-06T17:20:12.781Z",253 "bumped": true,254 "bumped_at": "2025-01-06T17:20:12.781Z",255 "archetype": "regular",256 "unseen": false,257 "pinned": false,258 "unpinned": null,259 "visible": true,260 "closed": false,261 "archived": false,262 "bookmarked": null,263 "liked": null,264 "tags_descriptions": {},265 "like_count": 1,266 "views": 105,267 "category_id": 12,268 "featured_link": null,269 "has_accepted_answer": true,270 "posters": [271 {272 "extras": null,273 "description": "Original Poster, Accepted Answer",274 "user": {275 "id": 81545,276 "username": "SzymonOzog",277 "name": "Szymon Ożóg",278 "avatar_template": "/user_avatar/discuss.pytorch.org/szymonozog/{size}/74552_2.png",279 "trust_level": 1280 }281 },282 {283 "extras": "latest",284 "description": "Most Recent Poster",285 "user": {286 "id": 49515,287 "username": "agu",288 "name": "Andrew Gu",289 "avatar_template": "/user_avatar/discuss.pytorch.org/agu/{size}/49913_2.png",290 "trust_level": 2291 }292 }293 ]294 },295 {296 "fancy_title": "DDP training get slower than first few iteration",297 "id": 212204,298 "title": "DDP training get slower than first few iteration",299 "slug": "ddp-training-get-slower-than-first-few-iteration",300 "posts_count": 3,301 "reply_count": 1,302 "highest_post_number": 3,303 "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/e/9/e96971f81fe3add1a25239274255bf7d36886808.png",304 "created_at": "2024-10-28T11:27:39.122Z",305 "last_posted_at": "2024-11-01T05:47:29.441Z",306 "bumped": true,307 "bumped_at": "2024-11-01T05:47:29.441Z",308 "archetype": "regular",309 "unseen": false,310 "pinned": false,311 "unpinned": null,312 "visible": true,313 "closed": false,314 "archived": false,315 "bookmarked": null,316 "liked": null,317 "tags_descriptions": {},318 "like_count": 0,319 "views": 277,320 "category_id": 12,321 "featured_link": null,322 "has_accepted_answer": false,323 "posters": [324 {325 "extras": "latest",326 "description": "Original Poster, Most Recent Poster",327 "user": {328 "id": 80536,329 "username": "GEOLU",330 "name": "",331 "avatar_template": "/user_avatar/discuss.pytorch.org/geolu/{size}/73621_2.png",332 "trust_level": 1333 }334 },335 {336 "extras": null,337 "description": "Frequent Poster",338 "user": {339 "id": 3534,340 "username": "ptrblck",341 "name": "",342 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",343 "admin": true,344 "moderator": true,345 "trust_level": 2346 }347 }348 ]349 }350 ],351 "tags_descriptions": {},352 "fancy_title": "No ouput by running the pytorch DDP example",353 "id": 180658,354 "title": "No ouput by running the pytorch DDP example",355 "posts_count": 1,356 "created_at": "2023-05-25T04:45:01.758Z",357 "views": 233,358 "reply_count": 0,359 "like_count": 0,360 "last_posted_at": "2023-05-25T04:45:01.836Z",361 "visible": true,362 "closed": false,363 "archived": false,364 "has_summary": false,365 "archetype": "regular",366 "slug": "no-ouput-by-running-the-pytorch-ddp-example",367 "category_id": 12,368 "word_count": 119,369 "deleted_at": null,370 "user_id": 66450,371 "featured_link": null,372 "pinned_globally": false,373 "pinned_at": null,374 "pinned_until": null,375 "image_url": null,376 "slow_mode_seconds": 0,377 "draft": null,378 "draft_key": "topic_180658",379 "draft_sequence": null,380 "unpinned": null,381 "pinned": false,382 "current_post_number": 1,383 "highest_post_number": 1,384 "deleted_by": null,385 "actions_summary": [386 {387 "id": 4,388 "count": 0,389 "hidden": false,390 "can_act": false391 },392 {393 "id": 8,394 "count": 0,395 "hidden": false,396 "can_act": false397 },398 {399 "id": 10,400 "count": 0,401 "hidden": false,402 "can_act": false403 },404 {405 "id": 7,406 "count": 0,407 "hidden": false,408 "can_act": false409 }410 ],411 "chunk_size": 20,412 "bookmarked": false,413 "topic_timer": null,414 "message_bus_last_id": 0,415 "participant_count": 1,416 "show_read_indicator": false,417 "thumbnails": null,418 "slow_mode_enabled_until": null,419 "can_vote": false,420 "vote_count": 0,421 "user_voted": false,422 "discourse_zendesk_plugin_zendesk_id": null,423 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",424 "details": {425 "can_edit": false,426 "notification_level": 1,427 "participants": [428 {429 "id": 66450,430 "username": "Yihua_Xu",431 "name": "Yihua Xu",432 "avatar_template": "/user_avatar/discuss.pytorch.org/yihua_xu/{size}/59111_2.png",433 "post_count": 1,434 "primary_group_name": null,435 "flair_name": null,436 "flair_url": null,437 "flair_color": null,438 "flair_bg_color": null,439 "flair_group_id": null,440 "trust_level": 1441 }442 ],443 "created_by": {444 "id": 66450,445 "username": "Yihua_Xu",446 "name": "Yihua Xu",447 "avatar_template": "/user_avatar/discuss.pytorch.org/yihua_xu/{size}/59111_2.png"448 },449 "last_poster": {450 "id": 66450,451 "username": "Yihua_Xu",452 "name": "Yihua Xu",453 "avatar_template": "/user_avatar/discuss.pytorch.org/yihua_xu/{size}/59111_2.png"454 },455 "links": [456 {457 "url": "https://github.com/pytorch/examples/blob/main/distributed/ddp-tutorial-series/multigpu.py",458 "title": "examples/multigpu.py at main · pytorch/examples · GitHub",459 "internal": false,460 "attachment": false,461 "reflection": false,462 "clicks": 1,463 "user_id": 66450,464 "domain": "github.com",465 "root_domain": "github.com"466 }467 ]468 },469 "bookmarks": []470 },471 {472 "post_stream": {473 "posts": [474 {475 "id": 403051,476 "name": "Marco",477 "username": "marco_c",478 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",479 "created_at": "2023-05-22T19:59:35.539Z",480 "cooked": "<p>Hello everybody! Could you please provide me neural network architecture suggestion(s) for video prediction (image sequence) regarding entering 144 images and predicting 48 images for each sequence? For illustration, I tested the Wavenet and CNNLSTM networks contained in <a href=\"https://bitbucket.org/retiarus/tec_prediction/src/conv-lstm/model/networks.py\" class=\"inline-onebox\" rel=\"noopener nofollow ugc\">Bitbucket</a> (but not only them) and I didn’t get good results. Thanks in advance!</p>",481 "post_number": 1,482 "post_type": 1,483 "posts_count": 12,484 "updated_at": "2023-05-23T00:52:32.920Z",485 "reply_count": 0,486 "reply_to_post_number": null,487 "quote_count": 0,488 "incoming_link_count": 599,489 "reads": 15,490 "readers_count": 14,491 "score": 2983.0,492 "yours": false,493 "topic_id": 180484,494 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",495 "display_username": "Marco",496 "primary_group_name": null,497 "flair_name": null,498 "flair_url": null,499 "flair_bg_color": null,500 "flair_color": null,501 "flair_group_id": null,502 "badges_granted": [],503 "version": 3,504 "can_edit": false,505 "can_delete": false,506 "can_recover": false,507 "can_see_hidden_post": false,508 "can_wiki": false,509 "link_counts": [510 {511 "url": "https://bitbucket.org/retiarus/tec_prediction/src/conv-lstm/model/networks.py",512 "internal": false,513 "reflection": false,514 "title": "Bitbucket",515 "clicks": 4516 }517 ],518 "read": true,519 "user_title": null,520 "bookmarked": false,521 "actions_summary": [],522 "moderator": false,523 "admin": false,524 "staff": false,525 "user_id": 62688,526 "hidden": false,527 "trust_level": 1,528 "deleted_at": null,529 "user_deleted": false,530 "edit_reason": null,531 "can_view_edit_history": true,532 "wiki": false,533 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/1",534 "can_accept_answer": false,535 "can_unaccept_answer": false,536 "accepted_answer": false,537 "topic_accepted_answer": null,538 "can_vote": false539 },540 {541 "id": 403093,542 "name": "Kapil Rana",543 "username": "Kapil_Rana",544 "avatar_template": "/user_avatar/discuss.pytorch.org/kapil_rana/{size}/20951_2.png",545 "created_at": "2023-05-23T05:40:21.568Z",546 "cooked": "<p>If you have good computation resources, You can try <strong>vision transformers</strong>. There are multiple papers in the literature.</p>",547 "post_number": 2,548 "post_type": 1,549 "posts_count": 12,550 "updated_at": "2023-05-23T05:40:21.568Z",551 "reply_count": 1,552 "reply_to_post_number": null,553 "quote_count": 0,554 "incoming_link_count": 6,555 "reads": 15,556 "readers_count": 14,557 "score": 38.0,558 "yours": false,559 "topic_id": 180484,560 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",561 "display_username": "Kapil Rana",562 "primary_group_name": null,563 "flair_name": null,564 "flair_url": null,565 "flair_bg_color": null,566 "flair_color": null,567 "flair_group_id": null,568 "badges_granted": [],569 "version": 1,570 "can_edit": false,571 "can_delete": false,572 "can_recover": false,573 "can_see_hidden_post": false,574 "can_wiki": false,575 "read": true,576 "user_title": null,577 "bookmarked": false,578 "actions_summary": [],579 "moderator": false,580 "admin": false,581 "staff": false,582 "user_id": 28082,583 "hidden": false,584 "trust_level": 2,585 "deleted_at": null,586 "user_deleted": false,587 "edit_reason": null,588 "can_view_edit_history": true,589 "wiki": false,590 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/2",591 "can_accept_answer": false,592 "can_unaccept_answer": false,593 "accepted_answer": false,594 "topic_accepted_answer": null595 },596 {597 "id": 403115,598 "name": "J Johnson",599 "username": "J_Johnson",600 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",601 "created_at": "2023-05-23T09:14:32.091Z",602 "cooked": "<p>How I’d approach this is basically modify a 2d UNet to use all 3d layers, and possibly adding in a 4d branch internally to the model, to assist with modeling 3d objects/scenes as they change over time(3d+1d=4d).</p>\n<p>Here is an example of a standard 2d Unet as well as a 2d + 3d Unet hybrid:</p>\n<aside class=\"onebox allowlistedgeneric\" data-onebox-src=\"https://github.com/CerebralSeed/Hybrid-3D-UNet\">\n <header class=\"source\">\n <img src=\"https://github.githubassets.com/favicons/favicon.svg\" class=\"site-icon\" width=\"32\" height=\"32\">\n\n <a href=\"https://github.com/CerebralSeed/Hybrid-3D-UNet\" target=\"_blank\" rel=\"noopener nofollow ugc\">GitHub</a>\n </header>\n\n <article class=\"onebox-body\">\n <div class=\"aspect-image\" style=\"--aspect-ratio:690/345;\"><img src=\"https://opengraph.githubassets.com/50c6e1f73a90b1bd45c337894e702bfcc67ceb2e2b0bdba33685617f4819bc6f/CerebralSeed/Hybrid-3D-UNet\" class=\"thumbnail\" width=\"690\" height=\"345\"></div>\n\n<h3><a href=\"https://github.com/CerebralSeed/Hybrid-3D-UNet\" target=\"_blank\" rel=\"noopener nofollow ugc\">GitHub - CerebralSeed/Hybrid-3D-UNet: Model for Hybrid 3D UNet</a></h3>\n\n <p>Model for Hybrid 3D UNet. Contribute to CerebralSeed/Hybrid-3D-UNet development by creating an account on GitHub.</p>\n\n\n </article>\n\n <div class=\"onebox-metadata\">\n \n \n </div>\n\n <div style=\"clear: both\"></div>\n</aside>\n\n<p>You’ll want to ensure your “channels” are still RBG and not the time sequence dimension. Also, adjust the kernel sizes throughout so it accepts 144 into the encoder and gives 48 out on the decoder on the time sequence dim.</p>",603 "post_number": 3,604 "post_type": 1,605 "posts_count": 12,606 "updated_at": "2023-05-23T09:14:32.091Z",607 "reply_count": 1,608 "reply_to_post_number": null,609 "quote_count": 0,610 "incoming_link_count": 4,611 "reads": 15,612 "readers_count": 14,613 "score": 28.0,614 "yours": false,615 "topic_id": 180484,616 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",617 "display_username": "J Johnson",618 "primary_group_name": null,619 "flair_name": null,620 "flair_url": null,621 "flair_bg_color": null,622 "flair_color": null,623 "flair_group_id": null,624 "badges_granted": [],625 "version": 1,626 "can_edit": false,627 "can_delete": false,628 "can_recover": false,629 "can_see_hidden_post": false,630 "can_wiki": false,631 "link_counts": [632 {633 "url": "https://github.com/CerebralSeed/Hybrid-3D-UNet",634 "internal": false,635 "reflection": false,636 "title": "GitHub - CerebralSeed/Hybrid-3D-UNet: Model for Hybrid 3D UNet",637 "clicks": 37638 }639 ],640 "read": true,641 "user_title": null,642 "bookmarked": false,643 "actions_summary": [],644 "moderator": false,645 "admin": false,646 "staff": false,647 "user_id": 41458,648 "hidden": false,649 "trust_level": 2,650 "deleted_at": null,651 "user_deleted": false,652 "edit_reason": null,653 "can_view_edit_history": true,654 "wiki": false,655 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/3",656 "can_accept_answer": false,657 "can_unaccept_answer": false,658 "accepted_answer": false,659 "topic_accepted_answer": null660 },661 {662 "id": 403162,663 "name": "Marco",664 "username": "marco_c",665 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",666 "created_at": "2023-05-23T15:18:52.329Z",667 "cooked": "<p>Hi. Ok, I’ll look into that, thanks!</p>",668 "post_number": 4,669 "post_type": 1,670 "posts_count": 12,671 "updated_at": "2023-05-23T15:18:52.329Z",672 "reply_count": 0,673 "reply_to_post_number": 2,674 "quote_count": 0,675 "incoming_link_count": 8,676 "reads": 13,677 "readers_count": 12,678 "score": 42.6,679 "yours": false,680 "topic_id": 180484,681 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",682 "display_username": "Marco",683 "primary_group_name": null,684 "flair_name": null,685 "flair_url": null,686 "flair_bg_color": null,687 "flair_color": null,688 "flair_group_id": null,689 "badges_granted": [],690 "version": 1,691 "can_edit": false,692 "can_delete": false,693 "can_recover": false,694 "can_see_hidden_post": false,695 "can_wiki": false,696 "read": true,697 "user_title": null,698 "reply_to_user": {699 "id": 28082,700 "username": "Kapil_Rana",701 "name": "Kapil Rana",702 "avatar_template": "/user_avatar/discuss.pytorch.org/kapil_rana/{size}/20951_2.png"703 },704 "bookmarked": false,705 "actions_summary": [],706 "moderator": false,707 "admin": false,708 "staff": false,709 "user_id": 62688,710 "hidden": false,711 "trust_level": 1,712 "deleted_at": null,713 "user_deleted": false,714 "edit_reason": null,715 "can_view_edit_history": true,716 "wiki": false,717 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/4",718 "can_accept_answer": false,719 "can_unaccept_answer": false,720 "accepted_answer": false,721 "topic_accepted_answer": null722 },723 {724 "id": 403164,725 "name": "Marco",726 "username": "marco_c",727 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",728 "created_at": "2023-05-23T15:26:27.584Z",729 "cooked": "<p>Hi. Ok, I took a look into that…</p>\n<p>I’m also currently testing a UNET-LSTM network according to the code below, but it also didn’t show good results so far, it would be very different from the UNET you suggested, besides the fact that it doesn’t have LSTM, right?</p>\n<pre><code class=\"lang-auto\"> def __init__(self, in_channels, out_channels):\n super().__init__()\n self.double_conv = nn.Sequential(\n nn.Conv3d(in_channels, out_channels, kernel_size=(1, 3, 3), padding=(0, 1, 1)),\n nn.BatchNorm3d(out_channels),\n nn.ReLU(inplace=True),\n nn.Conv3d(out_channels, out_channels, kernel_size=(1, 3, 3), padding=(0, 1, 1)),\n nn.BatchNorm3d(out_channels),\n nn.ReLU(inplace=True)\n )\n\n def forward(self, x):\n return self.double_conv(x)\n\nclass UnetLSTM(nn.Module):\n def __init__(self, in_channels, out_channels, num_layers, kernel_size, dilation, stride, dropout):\n super(UnetLSTM, self).__init__()\n\n self.in_channels = in_channels # Corresponds to input size\n self.out_channels = out_channels # Corresponds to hidden size\n self.num_layers = num_layers \n self.kernel_size = kernel_size\n self.dilation = dilation\n self.stride = stride\n self.dropout = dropout\n\n #self.conv1 = nn.Conv3d(in_channels=1, out_channels=128, kernel_size=3, stride=1, padding=1)\n #self.conv2 = nn.Conv3d(in_channels=128, out_channels=256, kernel_size=3, stride=1, padding=1)\n\n self.down1 = DoubleConv(in_channels, 64)\n self.pool1 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n self.down2 = DoubleConv(64, 128)\n self.pool2 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n #self.down3 = DoubleConv(128, 256)\n #self.pool3 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n #self.down4 = DoubleConv(256, 512)\n #self.pool4 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n\n #self.lstm1 = nn.LSTM(out_channels, hidden_channels, num_layers=num_layers, batch_first=True)\n #self.lstm2 = nn.LSTM(out_channels, hidden_channels, num_layers=num_layers, batch_first=True)\n #self.lstm3 = nn.LSTM(out_channels, hidden_channels, num_layers=num_layers, batch_first=True)\n \n self.lstm2 = crnn.Conv2dLSTM(in_channels=128, \n out_channels=128, \n kernel_size=self.kernel_size, \n num_layers=self.num_layers, \n bidirectional=True,\n dilation=self.dilation, \n stride=self.stride, \n dropout=self.dropout, \n batch_first=True) \n\n self.conv2 = nn.Conv3d(128*2, 128, kernel_size=1)\n\n self.up1 = nn.ConvTranspose3d(512, 256, kernel_size=(1, 2, 2), stride=(1, 2, 2))\n self.up_conv1 = DoubleConv(512, 256)\n self.up2 = nn.ConvTranspose3d(256, 128, kernel_size=(1, 2, 2), stride=(1, 2, 2))\n self.up_conv2 = DoubleConv(256, 128)\n self.up3 = nn.ConvTranspose3d(128, 64, kernel_size=(1, 2, 2), stride=(1, 2, 2), output_padding=(0, 1, 1))\n self.up_conv3 = DoubleConv(128, 64)\n self.out_conv = nn.Conv3d(64, out_channels, kernel_size=1)\n\n def forward(self, x):\n #x = self.conv1(x)\n #print(x.shape)\n #x = self.conv2(x)\n \n # Encoder\n x1 = self.down1(x)\n x2 = self.pool1(x1)\n x2 = self.down2(x2)\n #x3 = self.pool2(x2)\n #x3 = self.down3(x3)\n #x4 = self.pool3(x3)\n #x4 = self.down4(x4)\n\n # LSTM \n\n x2 = x2.permute(0, 2, 1, 3, 4)\n x2, _ = self.lstm2(x2)\n x2 = x2.permute(0, 2, 1, 3, 4)\n x2 = self.conv2(x2)\n\n # Decoder\n #x = self.up1(x4)\n #x = torch.cat([x, x3], dim=1)\n #x = self.up_conv1(x)\n\n #x = self.up2(x)\n #x = torch.cat([x, x2], dim=1)\n #x = self.up_conv2(x)\n\n x = self.up3(x2)\n x = torch.cat([x, x1], dim=1)\n x = self.up_conv3(x)\n \n x = self.out_conv(x) \n x = x.reshape(x.size(0), x.size(1), -1, x.size(3), x.size(4))\n \n #return x \n return x[:, :, -48:, :, :] \n</code></pre>\n<p>My images always have one channel (pixel values range from 0 to 255).<br>\nBy the way, do you see problems in this code I showed?</p>\n<p>Thank you very much!</p>",730 "post_number": 5,731 "post_type": 1,732 "posts_count": 12,733 "updated_at": "2023-05-23T15:30:08.829Z",734 "reply_count": 1,735 "reply_to_post_number": 3,736 "quote_count": 0,737 "incoming_link_count": 2,738 "reads": 11,739 "readers_count": 10,740 "score": 17.2,741 "yours": false,742 "topic_id": 180484,743 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",744 "display_username": "Marco",745 "primary_group_name": null,746 "flair_name": null,747 "flair_url": null,748 "flair_bg_color": null,749 "flair_color": null,750 "flair_group_id": null,751 "badges_granted": [],752 "version": 2,753 "can_edit": false,754 "can_delete": false,755 "can_recover": false,756 "can_see_hidden_post": false,757 "can_wiki": false,758 "read": true,759 "user_title": null,760 "reply_to_user": {761 "id": 41458,762 "username": "J_Johnson",763 "name": "J Johnson",764 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"765 },766 "bookmarked": false,767 "actions_summary": [],768 "moderator": false,769 "admin": false,770 "staff": false,771 "user_id": 62688,772 "hidden": false,773 "trust_level": 1,774 "deleted_at": null,775 "user_deleted": false,776 "edit_reason": null,777 "can_view_edit_history": true,778 "wiki": false,779 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/5",780 "can_accept_answer": false,781 "can_unaccept_answer": false,782 "accepted_answer": false,783 "topic_accepted_answer": null784 },785 {786 "id": 403189,787 "name": "J Johnson",788 "username": "J_Johnson",789 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",790 "created_at": "2023-05-23T17:45:58.743Z",791 "cooked": "<p>I’m not at a computer, so unable to test, but will point a few things I noticed.</p>\n<ol>\n<li>Should be 2d UNet LSTM or 3d UNet. The 3d Unet eliminates the need for an LSTM since you’re feeding the entire time sequence at once(albeit more calculation intensive).</li>\n<li>Self attention has good success in UNets. It helps the model focus on what’s important. Especially helps with skip connections, which I see you’ve included.</li>\n</ol>\n<p>Have you applied normalization between 0 and 1 for the inputs?</p>",792 "post_number": 6,793 "post_type": 1,794 "posts_count": 12,795 "updated_at": "2023-05-23T17:45:58.743Z",796 "reply_count": 1,797 "reply_to_post_number": 5,798 "quote_count": 0,799 "incoming_link_count": 1,800 "reads": 8,801 "readers_count": 7,802 "score": 11.6,803 "yours": false,804 "topic_id": 180484,805 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",806 "display_username": "J Johnson",807 "primary_group_name": null,808 "flair_name": null,809 "flair_url": null,810 "flair_bg_color": null,811 "flair_color": null,812 "flair_group_id": null,813 "badges_granted": [],814 "version": 1,815 "can_edit": false,816 "can_delete": false,817 "can_recover": false,818 "can_see_hidden_post": false,819 "can_wiki": false,820 "read": true,821 "user_title": null,822 "reply_to_user": {823 "id": 62688,824 "username": "marco_c",825 "name": "Marco",826 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png"827 },828 "bookmarked": false,829 "actions_summary": [],830 "moderator": false,831 "admin": false,832 "staff": false,833 "user_id": 41458,834 "hidden": false,835 "trust_level": 2,836 "deleted_at": null,837 "user_deleted": false,838 "edit_reason": null,839 "can_view_edit_history": true,840 "wiki": false,841 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/6",842 "can_accept_answer": false,843 "can_unaccept_answer": false,844 "accepted_answer": false,845 "topic_accepted_answer": null846 },847 {848 "id": 403201,849 "name": "Marco",850 "username": "marco_c",851 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",852 "created_at": "2023-05-23T18:44:40.544Z",853 "cooked": "<p>Ok, considering I’m using 3D UNET and the input format I’m using is [batch_size=x, channels=1, size=144, width=7, height=7], would I be able to adapt my code to use 2D UNET to continue using LSTM?</p>\n<p>Yes, I am normalizing the data into 0-1.</p>\n<p>About self attention, thanks for the tip, I’ll add that.</p>\n<p>Thank you very much!</p>",854 "post_number": 7,855 "post_type": 1,856 "posts_count": 12,857 "updated_at": "2023-05-23T18:52:49.349Z",858 "reply_count": 1,859 "reply_to_post_number": 6,860 "quote_count": 0,861 "incoming_link_count": 0,862 "reads": 6,863 "readers_count": 5,864 "score": 6.2,865 "yours": false,866 "topic_id": 180484,867 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",868 "display_username": "Marco",869 "primary_group_name": null,870 "flair_name": null,871 "flair_url": null,872 "flair_bg_color": null,873 "flair_color": null,874 "flair_group_id": null,875 "badges_granted": [],876 "version": 2,877 "can_edit": false,878 "can_delete": false,879 "can_recover": false,880 "can_see_hidden_post": false,881 "can_wiki": false,882 "read": true,883 "user_title": null,884 "reply_to_user": {885 "id": 41458,886 "username": "J_Johnson",887 "name": "J Johnson",888 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"889 },890 "bookmarked": false,891 "actions_summary": [],892 "moderator": false,893 "admin": false,894 "staff": false,895 "user_id": 62688,896 "hidden": false,897 "trust_level": 1,898 "deleted_at": null,899 "user_deleted": false,900 "edit_reason": null,901 "can_view_edit_history": true,902 "wiki": false,903 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/7",904 "can_accept_answer": false,905 "can_unaccept_answer": false,906 "accepted_answer": false,907 "topic_accepted_answer": null908 },909 {910 "id": 403259,911 "name": "J Johnson",912 "username": "J_Johnson",913 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",914 "created_at": "2023-05-24T06:13:05.084Z",915 "cooked": "<p>A 3D UNet or Vision Transformer would be superior to an RNN such as an LSTM.</p>\n<p>Consider this, what is the highest accuracy you’ve seen a model attain? The LSTM has to learn what information to pass on and what to discard. Let’s suppose, after much training, it does so correctly 80% of the time. That is an 80% memory accuracy(which for an ML model is pretty good). That means by the 4th time sequence, you’ve lost upward of 59% (1-0.8^4) fidelity.</p>\n<p>This is why language models before 2017 were not very good, compared to today. Why settle for 80% memory capture between time frames when you can just give a model the data at 100% fidelity?</p>\n<p>By passing in all time frames, as with a 3D UNet, you no longer need that middle LSTM layer. You can pass the data directly from encoder to decoder.</p>",916 "post_number": 8,917 "post_type": 1,918 "posts_count": 12,919 "updated_at": "2023-05-24T06:13:05.084Z",920 "reply_count": 1,921 "reply_to_post_number": 7,922 "quote_count": 0,923 "incoming_link_count": 4,924 "reads": 6,925 "readers_count": 5,926 "score": 26.2,927 "yours": false,928 "topic_id": 180484,929 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",930 "display_username": "J Johnson",931 "primary_group_name": null,932 "flair_name": null,933 "flair_url": null,934 "flair_bg_color": null,935 "flair_color": null,936 "flair_group_id": null,937 "badges_granted": [],938 "version": 1,939 "can_edit": false,940 "can_delete": false,941 "can_recover": false,942 "can_see_hidden_post": false,943 "can_wiki": false,944 "read": true,945 "user_title": null,946 "reply_to_user": {947 "id": 62688,948 "username": "marco_c",949 "name": "Marco",950 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png"951 },952 "bookmarked": false,953 "actions_summary": [],954 "moderator": false,955 "admin": false,956 "staff": false,957 "user_id": 41458,958 "hidden": false,959 "trust_level": 2,960 "deleted_at": null,961 "user_deleted": false,962 "edit_reason": null,963 "can_view_edit_history": true,964 "wiki": false,965 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/8",966 "can_accept_answer": false,967 "can_unaccept_answer": false,968 "accepted_answer": false,969 "topic_accepted_answer": null970 },971 {972 "id": 403318,973 "name": "Marco",974 "username": "marco_c",975 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",976 "created_at": "2023-05-24T15:53:58.018Z",977 "cooked": "<p>Hi! Right, I’m testing now the 3D Unet without the LSTM. I’m using this code, what do you think?</p>\n<pre><code class=\"lang-auto\">class DoubleConv(nn.Module):\n def __init__(self, in_channels, out_channels):\n super().__init__()\n self.double_conv = nn.Sequential(\n nn.Conv3d(in_channels, out_channels, kernel_size=(1, 3, 3), padding=(0, 1, 1)),\n nn.BatchNorm3d(out_channels),\n nn.ReLU(inplace=True),\n nn.Conv3d(out_channels, out_channels, kernel_size=(1, 3, 3), padding=(0, 1, 1)),\n nn.BatchNorm3d(out_channels),\n nn.ReLU(inplace=True)\n )\n\n def forward(self, x):\n return self.double_conv(x)\n\nclass Unet(nn.Module):\n def __init__(self, in_channels, out_channels):\n super(Unet, self).__init__()\n\n self.in_channels = in_channels\n self.out_channels = out_channels\n\n self.down1 = DoubleConv(in_channels, 64)\n self.pool1 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n self.down2 = DoubleConv(64, 128)\n self.pool2 = nn.MaxPool3d(kernel_size=(1, 2, 2))\n self.conv2 = nn.Conv3d(128, 128, kernel_size=1)\n self.up3 = nn.ConvTranspose3d(128, 64, kernel_size=(1, 2, 2), stride=(1, 2, 2), output_padding=(0, 1, 1))\n self.up_conv3 = DoubleConv(128, 64)\n self.out_conv = nn.Conv3d(64, out_channels, kernel_size=1)\n\n def forward(self, x):\n x1 = self.down1(x)\n x2 = self.pool1(x1)\n x2 = self.down2(x2)\n x2 = self.conv2(x2)\n x = self.up3(x2)\n x = torch.cat([x, x1], dim=1)\n x = self.up_conv3(x)\n x = self.out_conv(x)\n x = x.reshape(x.size(0), x.size(1), -1, x.size(3), x.size(4))\n return x[:, :, -48:, :, :]\n\n</code></pre>\n<p>Thank you so much for all the information and help you have given me!</p>",978 "post_number": 9,979 "post_type": 1,980 "posts_count": 12,981 "updated_at": "2023-05-24T15:53:58.018Z",982 "reply_count": 1,983 "reply_to_post_number": 8,984 "quote_count": 0,985 "incoming_link_count": 2,986 "reads": 6,987 "readers_count": 5,988 "score": 16.2,989 "yours": false,990 "topic_id": 180484,991 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",992 "display_username": "Marco",993 "primary_group_name": null,994 "flair_name": null,995 "flair_url": null,996 "flair_bg_color": null,997 "flair_color": null,998 "flair_group_id": null,999 "badges_granted": [],1000 "version": 1,1001 "can_edit": false,1002 "can_delete": false,1003 "can_recover": false,1004 "can_see_hidden_post": false,1005 "can_wiki": false,1006 "read": true,1007 "user_title": null,1008 "reply_to_user": {1009 "id": 41458,1010 "username": "J_Johnson",1011 "name": "J Johnson",1012 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"1013 },1014 "bookmarked": false,1015 "actions_summary": [],1016 "moderator": false,1017 "admin": false,1018 "staff": false,1019 "user_id": 62688,1020 "hidden": false,1021 "trust_level": 1,1022 "deleted_at": null,1023 "user_deleted": false,1024 "edit_reason": null,1025 "can_view_edit_history": true,1026 "wiki": false,1027 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/9",1028 "can_accept_answer": false,1029 "can_unaccept_answer": false,1030 "accepted_answer": false,1031 "topic_accepted_answer": null1032 },1033 {1034 "id": 403322,1035 "name": "J Johnson",1036 "username": "J_Johnson",1037 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",1038 "created_at": "2023-05-24T16:16:08.553Z",1039 "cooked": "<p>That looks better.</p>\n<p>I noticed you applied a skip connection near the beginning of the decoder but nowhere else. I’m guessing that was due to sizing issues since your output size is not the same as your input size. One way you can work around that is by just running a given skip tensor from the encoder through an nn.AdaptiveMaxPool3d or nn.AdaptiveAvgPool3d layer to get dim=2 to the right size needed. That will allow you to still pass those skip tensors for better fidelity when much of a scene remains the same in the output as it is in the input.</p>",1040 "post_number": 10,1041 "post_type": 1,1042 "posts_count": 12,1043 "updated_at": "2023-05-24T16:16:08.553Z",1044 "reply_count": 1,1045 "reply_to_post_number": 9,1046 "quote_count": 0,1047 "incoming_link_count": 5,1048 "reads": 6,1049 "readers_count": 5,1050 "score": 31.2,1051 "yours": false,1052 "topic_id": 180484,1053 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",1054 "display_username": "J Johnson",1055 "primary_group_name": null,1056 "flair_name": null,1057 "flair_url": null,1058 "flair_bg_color": null,1059 "flair_color": null,1060 "flair_group_id": null,1061 "badges_granted": [],1062 "version": 1,1063 "can_edit": false,1064 "can_delete": false,1065 "can_recover": false,1066 "can_see_hidden_post": false,1067 "can_wiki": false,1068 "read": true,1069 "user_title": null,1070 "reply_to_user": {1071 "id": 62688,1072 "username": "marco_c",1073 "name": "Marco",1074 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png"1075 },1076 "bookmarked": false,1077 "actions_summary": [],1078 "moderator": false,1079 "admin": false,1080 "staff": false,1081 "user_id": 41458,1082 "hidden": false,1083 "trust_level": 2,1084 "deleted_at": null,1085 "user_deleted": false,1086 "edit_reason": null,1087 "can_view_edit_history": true,1088 "wiki": false,1089 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/10",1090 "can_accept_answer": false,1091 "can_unaccept_answer": false,1092 "accepted_answer": false,1093 "topic_accepted_answer": null1094 },1095 {1096 "id": 403328,1097 "name": "Marco",1098 "username": "marco_c",1099 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png",1100 "created_at": "2023-05-24T16:37:05.048Z",1101 "cooked": "<p>Hi. Ok, nice! About the skip connection, I think you’re talking about this line:</p>\n<pre><code class=\"lang-auto\">x = torch.cat([x, x1], dim=1)\n</code></pre>\n<p>I’m doing this just to allow low-level information captured in the first layers of the network to be preserved and combined with high-level information in subsequent layers, to help improve the model’s ability to capture both fine detail and more abstract level features. But I’ll try to do it the way you suggested.</p>\n<p>I made a version of the Vision Transformer network as below, what do you think? I haven’t tested it yet.</p>\n<pre><code class=\"lang-auto\">import torch\nimport torch.nn as nn\nfrom torch.nn import Transformer\n\nclass VisionTransformer(nn.Module):\n def __init__(self, in_channels, out_channels, seq_length=144, embed_dim=256, num_heads=8, num_layers=6):\n super(VisionTransformer, self).__init__()\n self.in_channels = in_channels\n self.out_channels = out_channels\n self.seq_length = seq_length\n self.embed_dim = embed_dim\n self.num_heads = num_heads\n self.num_layers = num_layers\n\n self.embedding = nn.Conv2d(in_channels, embed_dim, kernel_size=1)\n self.pos_embedding = nn.Parameter(torch.randn(1, seq_length, embed_dim))\n self.transformer = Transformer(\n d_model=embed_dim,\n nhead=num_heads,\n num_encoder_layers=num_layers,\n num_decoder_layers=num_layers\n )\n self.fc = nn.Linear(embed_dim, out_channels)\n self.output_conv = nn.Conv3d(out_channels, out_channels, kernel_size=1)\n\n def forward(self, x):\n bs, _, _, h, w = x.shape\n x = x.view(-1, self.in_channels, h, w)\n\n x = self.embedding(x)\n\n # Adjust pos_embedding size to match x\n pos_embedding = self.pos_embedding.repeat(bs, h, w, 1)\n\n x += pos_embedding\n\n x = x.permute(0, 2, 3, 1)\n x = x.view(bs, -1, self.embed_dim)\n\n x = self.transformer(x, x)\n\n x = x.view(bs, h, w, self.embed_dim)\n x = x.permute(0, 3, 1, 2)\n\n x = self.fc(x)\n\n x = x.view(bs, self.out_channels, -1, h, w)\n\n x = self.output_conv(x)\n\n x = x.view(bs, -1, h, w)\n\n return x\n</code></pre>\n<p>Thanks!</p>",1102 "post_number": 11,1103 "post_type": 1,1104 "posts_count": 12,1105 "updated_at": "2023-05-24T18:51:49.637Z",1106 "reply_count": 1,1107 "reply_to_post_number": 10,1108 "quote_count": 0,1109 "incoming_link_count": 8,1110 "reads": 6,1111 "readers_count": 5,1112 "score": 46.2,1113 "yours": false,1114 "topic_id": 180484,1115 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",1116 "display_username": "Marco",1117 "primary_group_name": null,1118 "flair_name": null,1119 "flair_url": null,1120 "flair_bg_color": null,1121 "flair_color": null,1122 "flair_group_id": null,1123 "badges_granted": [],1124 "version": 5,1125 "can_edit": false,1126 "can_delete": false,1127 "can_recover": false,1128 "can_see_hidden_post": false,1129 "can_wiki": false,1130 "read": true,1131 "user_title": null,1132 "reply_to_user": {1133 "id": 41458,1134 "username": "J_Johnson",1135 "name": "J Johnson",1136 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"1137 },1138 "bookmarked": false,1139 "actions_summary": [],1140 "moderator": false,1141 "admin": false,1142 "staff": false,1143 "user_id": 62688,1144 "hidden": false,1145 "trust_level": 1,1146 "deleted_at": null,1147 "user_deleted": false,1148 "edit_reason": null,1149 "can_view_edit_history": true,1150 "wiki": false,1151 "post_url": "/t/neural-network-architecture-suggestion-s-for-video-prediction-image-sequence/180484/11",1152 "can_accept_answer": false,1153 "can_unaccept_answer": false,1154 "accepted_answer": false,1155 "topic_accepted_answer": null1156 },1157 {1158 "id": 403385,1159 "name": "J Johnson",1160 "username": "J_Johnson",1161 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",1162 "created_at": "2023-05-25T03:40:31.942Z",1163 "cooked": "<p>Regarding ViTs, I’ve only seen them used well in classification type problems. Not in image generation. If you go that route, you may need to read up on a few papers with keywords ViT and image generation. From what I have read, you may need to add a positional encoder.</p>",1164 "post_number": 12,1165 "post_type": 1,1166 "posts_count": 12,1167 "updated_at": "2023-05-25T03:40:31.942Z",1168 "reply_count": 0,1169 "reply_to_post_number": 11,1170 "quote_count": 0,1171 "incoming_link_count": 3,1172 "reads": 5,1173 "readers_count": 4,1174 "score": 16.0,1175 "yours": false,1176 "topic_id": 180484,1177 "topic_slug": "neural-network-architecture-suggestion-s-for-video-prediction-image-sequence",1178 "display_username": "J Johnson",1179 "primary_group_name": null,1180 "flair_name": null,1181 "flair_url": null,1182 "flair_bg_color": null,1183 "flair_color": null,1184 "flair_group_id": null,1185 "badges_granted": [],1186 "version": 1,1187 "can_edit": false,1188 "can_delete": false,1189 "can_recover": false,1190 "can_see_hidden_post": false,1191 "can_wiki": false,1192 "read": true,1193 "user_title": null,1194 "reply_to_user": {1195 "id": 62688,1196 "username": "marco_c",1197 "name": "Marco",1198 "avatar_template": "/letter_avatar_proxy/v4/letter/m/a87d85/{size}.png"1199 },1200 "bookmarked": false,