Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 474107,7 "name": "Joey",8 "username": "Joey1",9 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png",10 "created_at": "2025-08-12T08:22:45.113Z",11 "cooked": "<p>Can i ask, now i’m already add mask predict from unet as channel 4 to image RGB, but my macro f1 still around 69%. Now I want to add more feature like age,location,… in fully connected.</p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 5,15 "updated_at": "2025-08-12T08:22:45.113Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 1,20 "reads": 15,21 "readers_count": 14,22 "score": 8.0,23 "yours": false,24 "topic_id": 222273,25 "topic_slug": "add-more-feature-in-last-layer",26 "display_username": "Joey",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 1,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "read": true,41 "user_title": null,42 "bookmarked": false,43 "actions_summary": [],44 "moderator": false,45 "admin": false,46 "staff": false,47 "user_id": 85462,48 "hidden": false,49 "trust_level": 1,50 "deleted_at": null,51 "user_deleted": false,52 "edit_reason": null,53 "can_view_edit_history": true,54 "wiki": false,55 "post_url": "/t/add-more-feature-in-last-layer/222273/1",56 "can_accept_answer": false,57 "can_unaccept_answer": false,58 "accepted_answer": false,59 "topic_accepted_answer": null,60 "can_vote": false61 },62 {63 "id": 474229,64 "name": "Joey",65 "username": "Joey1",66 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png",67 "created_at": "2025-08-14T03:04:52.947Z",68 "cooked": "<p>thank you so much for your support</p>",69 "post_number": 3,70 "post_type": 1,71 "posts_count": 5,72 "updated_at": "2025-08-14T03:04:52.947Z",73 "reply_count": 1,74 "reply_to_post_number": 2,75 "quote_count": 0,76 "incoming_link_count": 0,77 "reads": 12,78 "readers_count": 11,79 "score": 7.4,80 "yours": false,81 "topic_id": 222273,82 "topic_slug": "add-more-feature-in-last-layer",83 "display_username": "Joey",84 "primary_group_name": null,85 "flair_name": null,86 "flair_url": null,87 "flair_bg_color": null,88 "flair_color": null,89 "flair_group_id": null,90 "badges_granted": [],91 "version": 1,92 "can_edit": false,93 "can_delete": false,94 "can_recover": false,95 "can_see_hidden_post": false,96 "can_wiki": false,97 "read": true,98 "user_title": null,99 "bookmarked": false,100 "actions_summary": [],101 "moderator": false,102 "admin": false,103 "staff": false,104 "user_id": 85462,105 "hidden": false,106 "trust_level": 1,107 "deleted_at": null,108 "user_deleted": false,109 "edit_reason": null,110 "can_view_edit_history": true,111 "wiki": false,112 "post_url": "/t/add-more-feature-in-last-layer/222273/3",113 "can_accept_answer": false,114 "can_unaccept_answer": false,115 "accepted_answer": false,116 "topic_accepted_answer": null117 },118 {119 "id": 474232,120 "name": "J Johnson",121 "username": "J_Johnson",122 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",123 "created_at": "2025-08-14T05:14:51.847Z",124 "cooked": "<p>Welcome to the forums. There are a lot of unet models on GitHub. Can you link or copy your current model code?</p>",125 "post_number": 4,126 "post_type": 1,127 "posts_count": 5,128 "updated_at": "2025-08-14T05:14:51.847Z",129 "reply_count": 1,130 "reply_to_post_number": 3,131 "quote_count": 0,132 "incoming_link_count": 2,133 "reads": 11,134 "readers_count": 10,135 "score": 17.2,136 "yours": false,137 "topic_id": 222273,138 "topic_slug": "add-more-feature-in-last-layer",139 "display_username": "J Johnson",140 "primary_group_name": null,141 "flair_name": null,142 "flair_url": null,143 "flair_bg_color": null,144 "flair_color": null,145 "flair_group_id": null,146 "badges_granted": [],147 "version": 1,148 "can_edit": false,149 "can_delete": false,150 "can_recover": false,151 "can_see_hidden_post": false,152 "can_wiki": false,153 "read": true,154 "user_title": null,155 "reply_to_user": {156 "id": 85462,157 "username": "Joey1",158 "name": "Joey",159 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png"160 },161 "bookmarked": false,162 "actions_summary": [],163 "moderator": false,164 "admin": false,165 "staff": false,166 "user_id": 41458,167 "hidden": false,168 "trust_level": 2,169 "deleted_at": null,170 "user_deleted": false,171 "edit_reason": null,172 "can_view_edit_history": true,173 "wiki": false,174 "post_url": "/t/add-more-feature-in-last-layer/222273/4",175 "can_accept_answer": false,176 "can_unaccept_answer": false,177 "accepted_answer": false,178 "topic_accepted_answer": null179 },180 {181 "id": 474248,182 "name": "Joey",183 "username": "Joey1",184 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png",185 "created_at": "2025-08-14T10:00:41.514Z",186 "cooked": "<p>thanks, i can add more feature but my macro still around 85%. Can I ask that is there any other way to increase the macro of the model, now I’m predict benign or malignant using HAM10000.</p>",187 "post_number": 8,188 "post_type": 1,189 "posts_count": 5,190 "updated_at": "2025-08-14T10:00:41.514Z",191 "reply_count": 1,192 "reply_to_post_number": 4,193 "quote_count": 0,194 "incoming_link_count": 1,195 "reads": 11,196 "readers_count": 10,197 "score": 12.2,198 "yours": false,199 "topic_id": 222273,200 "topic_slug": "add-more-feature-in-last-layer",201 "display_username": "Joey",202 "primary_group_name": null,203 "flair_name": null,204 "flair_url": null,205 "flair_bg_color": null,206 "flair_color": null,207 "flair_group_id": null,208 "badges_granted": [],209 "version": 1,210 "can_edit": false,211 "can_delete": false,212 "can_recover": false,213 "can_see_hidden_post": false,214 "can_wiki": false,215 "read": true,216 "user_title": null,217 "reply_to_user": {218 "id": 41458,219 "username": "J_Johnson",220 "name": "J Johnson",221 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"222 },223 "bookmarked": false,224 "actions_summary": [],225 "moderator": false,226 "admin": false,227 "staff": false,228 "user_id": 85462,229 "hidden": false,230 "trust_level": 1,231 "deleted_at": null,232 "user_deleted": false,233 "edit_reason": null,234 "can_view_edit_history": true,235 "wiki": false,236 "post_url": "/t/add-more-feature-in-last-layer/222273/8",237 "can_accept_answer": false,238 "can_unaccept_answer": false,239 "accepted_answer": false,240 "topic_accepted_answer": null241 },242 {243 "id": 474441,244 "name": "J Johnson",245 "username": "J_Johnson",246 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",247 "created_at": "2025-08-19T08:49:24.676Z",248 "cooked": "<p>I’m not clear on what you’re asking or what model you’re using. Can you provide the code for your model or a sample snippet of code that reproduces the issue?</p>",249 "post_number": 9,250 "post_type": 1,251 "posts_count": 5,252 "updated_at": "2025-08-19T08:49:24.676Z",253 "reply_count": 0,254 "reply_to_post_number": 8,255 "quote_count": 0,256 "incoming_link_count": 0,257 "reads": 11,258 "readers_count": 10,259 "score": 2.2,260 "yours": false,261 "topic_id": 222273,262 "topic_slug": "add-more-feature-in-last-layer",263 "display_username": "J Johnson",264 "primary_group_name": null,265 "flair_name": null,266 "flair_url": null,267 "flair_bg_color": null,268 "flair_color": null,269 "flair_group_id": null,270 "badges_granted": [],271 "version": 1,272 "can_edit": false,273 "can_delete": false,274 "can_recover": false,275 "can_see_hidden_post": false,276 "can_wiki": false,277 "read": true,278 "user_title": null,279 "reply_to_user": {280 "id": 85462,281 "username": "Joey1",282 "name": "Joey",283 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png"284 },285 "bookmarked": false,286 "actions_summary": [],287 "moderator": false,288 "admin": false,289 "staff": false,290 "user_id": 41458,291 "hidden": false,292 "trust_level": 2,293 "deleted_at": null,294 "user_deleted": false,295 "edit_reason": null,296 "can_view_edit_history": true,297 "wiki": false,298 "post_url": "/t/add-more-feature-in-last-layer/222273/9",299 "can_accept_answer": false,300 "can_unaccept_answer": false,301 "accepted_answer": false,302 "topic_accepted_answer": null303 }304 ],305 "stream": [306 474107,307 474229,308 474232,309 474248,310 474441311 ]312 },313 "timeline_lookup": [314 [315 1,316 74317 ],318 [319 2,320 73321 ],322 [323 4,324 72325 ],326 [327 5,328 67329 ]330 ],331 "suggested_topics": [332 {333 "fancy_title": "Extracting Swin-Vit backbone",334 "id": 212799,335 "title": "Extracting Swin-Vit backbone",336 "slug": "extracting-swin-vit-backbone",337 "posts_count": 1,338 "reply_count": 0,339 "highest_post_number": 1,340 "image_url": null,341 "created_at": "2024-11-11T09:54:48.491Z",342 "last_posted_at": "2024-11-11T09:54:48.540Z",343 "bumped": true,344 "bumped_at": "2024-11-11T09:54:48.540Z",345 "archetype": "regular",346 "unseen": false,347 "pinned": false,348 "unpinned": null,349 "visible": true,350 "closed": false,351 "archived": false,352 "bookmarked": null,353 "liked": null,354 "tags_descriptions": {},355 "like_count": 0,356 "views": 195,357 "category_id": 5,358 "featured_link": null,359 "has_accepted_answer": false,360 "posters": [361 {362 "extras": "latest single",363 "description": "Original Poster, Most Recent Poster",364 "user": {365 "id": 56688,366 "username": "ima",367 "name": "Imantha gunasekera",368 "avatar_template": "/letter_avatar_proxy/v4/letter/i/67e7ee/{size}.png",369 "trust_level": 1370 }371 }372 ]373 },374 {375 "fancy_title": "How can I implement operations on queue with supported MIL ops?",376 "id": 213763,377 "title": "How can I implement operations on queue with supported MIL ops?",378 "slug": "how-can-i-implement-operations-on-queue-with-supported-mil-ops",379 "posts_count": 1,380 "reply_count": 0,381 "highest_post_number": 1,382 "image_url": null,383 "created_at": "2024-12-03T22:35:04.894Z",384 "last_posted_at": "2024-12-03T22:35:04.945Z",385 "bumped": true,386 "bumped_at": "2024-12-03T22:35:04.945Z",387 "archetype": "regular",388 "unseen": false,389 "pinned": false,390 "unpinned": null,391 "visible": true,392 "closed": false,393 "archived": false,394 "bookmarked": null,395 "liked": null,396 "tags_descriptions": {},397 "like_count": 0,398 "views": 18,399 "category_id": 5,400 "featured_link": null,401 "has_accepted_answer": false,402 "posters": [403 {404 "extras": "latest single",405 "description": "Original Poster, Most Recent Poster",406 "user": {407 "id": 45116,408 "username": "JimW",409 "name": "",410 "avatar_template": "/user_avatar/discuss.pytorch.org/jimw/{size}/38000_2.png",411 "trust_level": 1412 }413 }414 ]415 },416 {417 "fancy_title": "Can I rotate two dimensional images (e.g. MNIST) without having to unsqueeze each one?",418 "id": 215191,419 "title": "Can I rotate two dimensional images (e.g. MNIST) without having to unsqueeze each one?",420 "slug": "can-i-rotate-two-dimensional-images-e-g-mnist-without-having-to-unsqueeze-each-one",421 "posts_count": 2,422 "reply_count": 0,423 "highest_post_number": 2,424 "image_url": null,425 "created_at": "2025-01-10T07:45:35.119Z",426 "last_posted_at": "2025-01-10T18:13:41.335Z",427 "bumped": true,428 "bumped_at": "2025-01-10T18:13:41.335Z",429 "archetype": "regular",430 "unseen": false,431 "pinned": false,432 "unpinned": null,433 "visible": true,434 "closed": false,435 "archived": false,436 "bookmarked": null,437 "liked": null,438 "tags_descriptions": {},439 "like_count": 1,440 "views": 54,441 "category_id": 5,442 "featured_link": null,443 "has_accepted_answer": true,444 "posters": [445 {446 "extras": null,447 "description": "Original Poster",448 "user": {449 "id": 68895,450 "username": "DawidL",451 "name": "",452 "avatar_template": "/user_avatar/discuss.pytorch.org/dawidl/{size}/62949_2.png",453 "trust_level": 1454 }455 },456 {457 "extras": "latest",458 "description": "Most Recent Poster, Accepted Answer",459 "user": {460 "id": 3534,461 "username": "ptrblck",462 "name": "",463 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",464 "admin": true,465 "moderator": true,466 "trust_level": 2467 }468 }469 ]470 },471 {472 "fancy_title": "F.scaled_dot_product_attention get query @ key",473 "id": 215697,474 "title": "F.scaled_dot_product_attention get query @ key",475 "slug": "f-scaled-dot-product-attention-get-query-key",476 "posts_count": 1,477 "reply_count": 0,478 "highest_post_number": 1,479 "image_url": null,480 "created_at": "2025-01-22T05:03:27.120Z",481 "last_posted_at": "2025-01-22T05:03:27.155Z",482 "bumped": true,483 "bumped_at": "2025-01-22T05:03:27.155Z",484 "archetype": "regular",485 "unseen": false,486 "pinned": false,487 "unpinned": null,488 "visible": true,489 "closed": false,490 "archived": false,491 "bookmarked": null,492 "liked": null,493 "tags_descriptions": {},494 "like_count": 0,495 "views": 109,496 "category_id": 5,497 "featured_link": null,498 "has_accepted_answer": false,499 "posters": [500 {501 "extras": "latest single",502 "description": "Original Poster, Most Recent Poster",503 "user": {504 "id": 82230,505 "username": "b10901187",506 "name": "閎凱 鍾",507 "avatar_template": "/user_avatar/discuss.pytorch.org/b10901187/{size}/75232_2.png",508 "trust_level": 1509 }510 }511 ]512 },513 {514 "fancy_title": "Parallel processing a RealESRGAN",515 "id": 217772,516 "title": "Parallel processing a RealESRGAN",517 "slug": "parallel-processing-a-realesrgan",518 "posts_count": 1,519 "reply_count": 0,520 "highest_post_number": 1,521 "image_url": null,522 "created_at": "2025-03-13T04:18:41.381Z",523 "last_posted_at": "2025-03-13T04:18:41.420Z",524 "bumped": true,525 "bumped_at": "2025-03-13T04:18:41.420Z",526 "archetype": "regular",527 "unseen": false,528 "pinned": false,529 "unpinned": null,530 "visible": true,531 "closed": false,532 "archived": false,533 "bookmarked": null,534 "liked": null,535 "tags_descriptions": {},536 "like_count": 0,537 "views": 24,538 "category_id": 5,539 "featured_link": null,540 "has_accepted_answer": false,541 "posters": [542 {543 "extras": "latest single",544 "description": "Original Poster, Most Recent Poster",545 "user": {546 "id": 83232,547 "username": "HAli",548 "name": "",549 "avatar_template": "/user_avatar/discuss.pytorch.org/hali/{size}/76127_2.png",550 "trust_level": 0551 }552 }553 ]554 }555 ],556 "tags_descriptions": {},557 "fancy_title": "Add more feature in last layer",558 "id": 222273,559 "title": "Add more feature in last layer",560 "posts_count": 5,561 "created_at": "2025-08-12T08:22:45.058Z",562 "views": 87,563 "reply_count": 5,564 "like_count": 0,565 "last_posted_at": "2025-08-19T08:49:24.676Z",566 "visible": true,567 "closed": false,568 "archived": false,569 "has_summary": false,570 "archetype": "regular",571 "slug": "add-more-feature-in-last-layer",572 "category_id": 5,573 "word_count": 137,574 "deleted_at": null,575 "user_id": 85462,576 "featured_link": null,577 "pinned_globally": false,578 "pinned_at": null,579 "pinned_until": null,580 "image_url": null,581 "slow_mode_seconds": 0,582 "draft": null,583 "draft_key": "topic_222273",584 "draft_sequence": null,585 "unpinned": null,586 "pinned": false,587 "current_post_number": 1,588 "highest_post_number": 9,589 "deleted_by": null,590 "actions_summary": [591 {592 "id": 4,593 "count": 0,594 "hidden": false,595 "can_act": false596 },597 {598 "id": 8,599 "count": 0,600 "hidden": false,601 "can_act": false602 },603 {604 "id": 10,605 "count": 0,606 "hidden": false,607 "can_act": false608 },609 {610 "id": 7,611 "count": 0,612 "hidden": false,613 "can_act": false614 }615 ],616 "chunk_size": 20,617 "bookmarked": false,618 "topic_timer": null,619 "message_bus_last_id": 0,620 "participant_count": 2,621 "show_read_indicator": false,622 "thumbnails": null,623 "slow_mode_enabled_until": null,624 "can_vote": false,625 "vote_count": 0,626 "user_voted": false,627 "discourse_zendesk_plugin_zendesk_id": null,628 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",629 "details": {630 "can_edit": false,631 "notification_level": 1,632 "participants": [633 {634 "id": 85462,635 "username": "Joey1",636 "name": "Joey",637 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png",638 "post_count": 3,639 "primary_group_name": null,640 "flair_name": null,641 "flair_url": null,642 "flair_color": null,643 "flair_bg_color": null,644 "flair_group_id": null,645 "trust_level": 1646 },647 {648 "id": 41458,649 "username": "J_Johnson",650 "name": "J Johnson",651 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",652 "post_count": 2,653 "primary_group_name": null,654 "flair_name": null,655 "flair_url": null,656 "flair_color": null,657 "flair_bg_color": null,658 "flair_group_id": null,659 "trust_level": 2660 }661 ],662 "created_by": {663 "id": 85462,664 "username": "Joey1",665 "name": "Joey",666 "avatar_template": "/user_avatar/discuss.pytorch.org/joey1/{size}/74683_2.png"667 },668 "last_poster": {669 "id": 41458,670 "username": "J_Johnson",671 "name": "J Johnson",672 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"673 }674 },675 "bookmarks": []676 },677 {678 "post_stream": {679 "posts": [680 {681 "id": 474316,682 "name": "Citystrawman",683 "username": "citystrawman",684 "avatar_template": "/user_avatar/discuss.pytorch.org/citystrawman/{size}/68437_2.png",685 "created_at": "2025-08-16T15:29:15.487Z",686 "cooked": "<p>I just started learning NLP and RNN. I read O’Reilly’s <a href=\"https://www.oreilly.com/library/view/zerokarazuo-rudeep-learning/9784873118369/\" rel=\"noopener nofollow ugc\">Deep Learning: natrual language processing</a>, and here’s the Figure showing batch-training in RNN:</p>\n<p><div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/3X/4/4/440da3b6c9137829f920dbb36869068895b48257.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/440da3b6c9137829f920dbb36869068895b48257\" title=\"RNN\"><img src=\"https://discuss.pytorch.org/uploads/default/optimized/3X/4/4/440da3b6c9137829f920dbb36869068895b48257_2_488x500.png\" alt=\"RNN\" data-base62-sha1=\"9I1GOhNccmQUgApCwboRXP9kVHV\" width=\"488\" height=\"500\" srcset=\"https://discuss.pytorch.org/uploads/default/optimized/3X/4/4/440da3b6c9137829f920dbb36869068895b48257_2_488x500.png, https://discuss.pytorch.org/uploads/default/optimized/3X/4/4/440da3b6c9137829f920dbb36869068895b48257_2_732x750.png 1.5x, https://discuss.pytorch.org/uploads/default/optimized/3X/4/4/440da3b6c9137829f920dbb36869068895b48257_2_976x1000.png 2x\" data-dominant-color=\"F5F7F8\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">RNN</span><span class=\"informations\">978×1002 61.6 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div></p>\n<p>(please ignore its Chinese Characters which are not important)</p>\n<p>In the above figure, the author uses Truncated BPTT as an example to show how to traing from a 1000-word time series, truncated at every 10 words, with batch size=2.</p>\n<p>As the figure shows, the first sequence in first batch from X0, and the second sequence in first batch start from X500, which shift from X0 by 500. So does the other batch.</p>\n<p>Then here comes my question: does this mean that all sequences in the first batch has no hidden state to inherit? For example, X500 does not have h499. What is more, does that mean that forward propagating is truncated at X500 (which indicates that the longest memory in this situation could only support 500 words)?</p>",687 "post_number": 1,688 "post_type": 1,689 "posts_count": 4,690 "updated_at": "2025-08-16T15:29:15.487Z",691 "reply_count": 0,692 "reply_to_post_number": null,693 "quote_count": 0,694 "incoming_link_count": 8,695 "reads": 6,696 "readers_count": 5,697 "score": 41.2,698 "yours": false,699 "topic_id": 222402,700 "topic_slug": "a-question-for-batch-training-rnn",701 "display_username": "Citystrawman",702 "primary_group_name": null,703 "flair_name": null,704 "flair_url": null,705 "flair_bg_color": null,706 "flair_color": null,707 "flair_group_id": null,708 "badges_granted": [],709 "version": 1,710 "can_edit": false,711 "can_delete": false,712 "can_recover": false,713 "can_see_hidden_post": false,714 "can_wiki": false,715 "link_counts": [716 {717 "url": "https://www.oreilly.com/library/view/zerokarazuo-rudeep-learning/9784873118369/",718 "internal": false,719 "reflection": false,720 "title": "ゼロから作るDeep Learning ❷ ―自然言語処理編 [Book]",721 "clicks": 2722 }723 ],724 "read": true,725 "user_title": null,726 "bookmarked": false,727 "actions_summary": [],728 "moderator": false,729 "admin": false,730 "staff": false,731 "user_id": 74088,732 "hidden": false,733 "trust_level": 0,734 "deleted_at": null,735 "user_deleted": false,736 "edit_reason": null,737 "can_view_edit_history": true,738 "wiki": false,739 "post_url": "/t/a-question-for-batch-training-rnn/222402/1",740 "can_accept_answer": false,741 "can_unaccept_answer": false,742 "accepted_answer": false,743 "topic_accepted_answer": null,744 "can_vote": false745 },746 {747 "id": 474409,748 "name": "J Johnson",749 "username": "J_Johnson",750 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",751 "created_at": "2025-08-18T16:19:43.276Z",752 "cooked": "<p>No. If you’re truncating, the first in the new truncated sequence gets zeros for the hidden state at t0, same as the initial sequence. Then that new hidden state moves on to t1, and so on.</p>",753 "post_number": 2,754 "post_type": 1,755 "posts_count": 4,756 "updated_at": "2025-08-18T16:19:43.276Z",757 "reply_count": 1,758 "reply_to_post_number": null,759 "quote_count": 0,760 "incoming_link_count": 1,761 "reads": 6,762 "readers_count": 5,763 "score": 11.2,764 "yours": false,765 "topic_id": 222402,766 "topic_slug": "a-question-for-batch-training-rnn",767 "display_username": "J Johnson",768 "primary_group_name": null,769 "flair_name": null,770 "flair_url": null,771 "flair_bg_color": null,772 "flair_color": null,773 "flair_group_id": null,774 "badges_granted": [],775 "version": 1,776 "can_edit": false,777 "can_delete": false,778 "can_recover": false,779 "can_see_hidden_post": false,780 "can_wiki": false,781 "read": true,782 "user_title": null,783 "bookmarked": false,784 "actions_summary": [],785 "moderator": false,786 "admin": false,787 "staff": false,788 "user_id": 41458,789 "hidden": false,790 "trust_level": 2,791 "deleted_at": null,792 "user_deleted": false,793 "edit_reason": null,794 "can_view_edit_history": true,795 "wiki": false,796 "post_url": "/t/a-question-for-batch-training-rnn/222402/2",797 "can_accept_answer": false,798 "can_unaccept_answer": false,799 "accepted_answer": false,800 "topic_accepted_answer": null801 },802 {803 "id": 474427,804 "name": "Citystrawman",805 "username": "citystrawman",806 "avatar_template": "/user_avatar/discuss.pytorch.org/citystrawman/{size}/68437_2.png",807 "created_at": "2025-08-19T01:52:56.788Z",808 "cooked": "<p>So, can I assume that if we use batch-training with batch number N, we just “separate” the whole texts into N parts which are not “connected“?</p>",809 "post_number": 3,810 "post_type": 1,811 "posts_count": 4,812 "updated_at": "2025-08-19T01:52:56.788Z",813 "reply_count": 1,814 "reply_to_post_number": 2,815 "quote_count": 0,816 "incoming_link_count": 0,817 "reads": 6,818 "readers_count": 5,819 "score": 6.2,820 "yours": false,821 "topic_id": 222402,822 "topic_slug": "a-question-for-batch-training-rnn",823 "display_username": "Citystrawman",824 "primary_group_name": null,825 "flair_name": null,826 "flair_url": null,827 "flair_bg_color": null,828 "flair_color": null,829 "flair_group_id": null,830 "badges_granted": [],831 "version": 1,832 "can_edit": false,833 "can_delete": false,834 "can_recover": false,835 "can_see_hidden_post": false,836 "can_wiki": false,837 "read": true,838 "user_title": null,839 "reply_to_user": {840 "id": 41458,841 "username": "J_Johnson",842 "name": "J Johnson",843 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png"844 },845 "bookmarked": false,846 "actions_summary": [],847 "moderator": false,848 "admin": false,849 "staff": false,850 "user_id": 74088,851 "hidden": false,852 "trust_level": 0,853 "deleted_at": null,854 "user_deleted": false,855 "edit_reason": null,856 "can_view_edit_history": true,857 "wiki": false,858 "post_url": "/t/a-question-for-batch-training-rnn/222402/3",859 "can_accept_answer": false,860 "can_unaccept_answer": false,861 "accepted_answer": false,862 "topic_accepted_answer": null863 },864 {865 "id": 474432,866 "name": "J Johnson",867 "username": "J_Johnson",868 "avatar_template": "/user_avatar/discuss.pytorch.org/j_johnson/{size}/55494_2.png",869 "created_at": "2025-08-19T02:50:54.214Z",870 "cooked": "<p>If you’re training set is one continuous document, then yes. But it’s probably a good idea to pad the beginning with a random amount of tokens between each epoch, in order to make those ‘cutoffs’ shifted and variable.</p>",871 "post_number": 4,872 "post_type": 1,873 "posts_count": 4,874 "updated_at": "2025-08-19T02:50:54.214Z",875 "reply_count": 0,876 "reply_to_post_number": 3,877 "quote_count": 0,878 "incoming_link_count": 1,879 "reads": 6,880 "readers_count": 5,881 "score": 21.2,882 "yours": false,883 "topic_id": 222402,884 "topic_slug": "a-question-for-batch-training-rnn",885 "display_username": "J Johnson",886 "primary_group_name": null,887 "flair_name": null,888 "flair_url": null,889 "flair_bg_color": null,890 "flair_color": null,891 "flair_group_id": null,892 "badges_granted": [],893 "version": 1,894 "can_edit": false,895 "can_delete": false,896 "can_recover": false,897 "can_see_hidden_post": false,898 "can_wiki": false,899 "read": true,900 "user_title": null,901 "reply_to_user": {902 "id": 74088,903 "username": "citystrawman",904 "name": "Citystrawman",905 "avatar_template": "/user_avatar/discuss.pytorch.org/citystrawman/{size}/68437_2.png"906 },907 "bookmarked": false,908 "actions_summary": [909 {910 "id": 2,911 "count": 1912 }913 ],914 "moderator": false,915 "admin": false,916 "staff": false,917 "user_id": 41458,918 "hidden": false,919 "trust_level": 2,920 "deleted_at": null,921 "user_deleted": false,922 "edit_reason": null,923 "can_view_edit_history": true,924 "wiki": false,925 "post_url": "/t/a-question-for-batch-training-rnn/222402/4",926 "can_accept_answer": false,927 "can_unaccept_answer": false,928 "accepted_answer": false,929 "topic_accepted_answer": null930 }931 ],932 "stream": [933 474316,934 474409,935 474427,936 474432937 ]938 },939 "timeline_lookup": [940 [941 1,942 70943 ],944 [945 2,946 68947 ]948 ],949 "suggested_topics": [950 {951 "fancy_title": "Pytorch OCR models for deploying to ESP32?",952 "id": 217755,953 "title": "Pytorch OCR models for deploying to ESP32?",954 "slug": "pytorch-ocr-models-for-deploying-to-esp32",955 "posts_count": 1,956 "reply_count": 0,957 "highest_post_number": 1,958 "image_url": null,959 "created_at": "2025-03-12T17:14:45.612Z",960 "last_posted_at": "2025-03-12T17:14:45.651Z",961 "bumped": true,962 "bumped_at": "2025-03-12T17:14:45.651Z",963 "archetype": "regular",964 "unseen": false,965 "pinned": false,966 "unpinned": null,967 "visible": true,968 "closed": false,969 "archived": false,970 "bookmarked": null,971 "liked": null,972 "tags_descriptions": {},973 "like_count": 0,974 "views": 126,975 "category_id": 8,976 "featured_link": null,977 "has_accepted_answer": false,978 "posters": [979 {980 "extras": "latest single",981 "description": "Original Poster, Most Recent Poster",982 "user": {983 "id": 83222,984 "username": "mavavilj",985 "name": "Matti",986 "avatar_template": "/user_avatar/discuss.pytorch.org/mavavilj/{size}/76119_2.png",987 "trust_level": 0988 }989 }990 ]991 },992 {993 "fancy_title": "Full finetune, LoRA and feature extraction take the same amount of memory and time to train",994 "id": 217833,995 "title": "Full finetune, LoRA and feature extraction take the same amount of memory and time to train",996 "slug": "full-finetune-lora-and-feature-extraction-take-the-same-amount-of-memory-and-time-to-train",997 "posts_count": 1,998 "reply_count": 0,999 "highest_post_number": 1,1000 "image_url": null,1001 "created_at": "2025-03-14T05:09:34.624Z",1002 "last_posted_at": "2025-03-14T05:09:34.664Z",1003 "bumped": true,1004 "bumped_at": "2025-03-14T05:40:12.747Z",1005 "archetype": "regular",1006 "unseen": false,1007 "pinned": false,1008 "unpinned": null,1009 "visible": true,1010 "closed": false,1011 "archived": false,1012 "bookmarked": null,1013 "liked": null,1014 "tags_descriptions": {},1015 "like_count": 0,1016 "views": 38,1017 "category_id": 8,1018 "featured_link": null,1019 "has_accepted_answer": false,1020 "posters": [1021 {1022 "extras": "latest single",1023 "description": "Original Poster, Most Recent Poster",1024 "user": {1025 "id": 75074,1026 "username": "Vefery",1027 "name": "",1028 "avatar_template": "/letter_avatar_proxy/v4/letter/v/d78d45/{size}.png",1029 "trust_level": 11030 }1031 }1032 ]1033 },1034 {1035 "fancy_title": "Correct way to batch custom masks in SDPA",1036 "id": 214155,1037 "title": "Correct way to batch custom masks in SDPA",1038 "slug": "correct-way-to-batch-custom-masks-in-sdpa",1039 "posts_count": 1,1040 "reply_count": 0,1041 "highest_post_number": 1,1042 "image_url": null,1043 "created_at": "2024-12-12T14:37:13.000Z",1044 "last_posted_at": "2024-12-12T14:37:13.049Z",1045 "bumped": true,1046 "bumped_at": "2024-12-12T14:37:13.049Z",1047 "archetype": "regular",1048 "unseen": false,1049 "pinned": false,1050 "unpinned": null,1051 "visible": true,1052 "closed": false,1053 "archived": false,1054 "bookmarked": null,1055 "liked": null,1056 "tags_descriptions": {},1057 "like_count": 0,1058 "views": 75,1059 "category_id": 8,1060 "featured_link": null,1061 "has_accepted_answer": false,1062 "posters": [1063 {1064 "extras": "latest single",1065 "description": "Original Poster, Most Recent Poster",1066 "user": {1067 "id": 81474,1068 "username": "mm23",1069 "name": "mm23",1070 "avatar_template": "/user_avatar/discuss.pytorch.org/mm23/{size}/74491_2.png",1071 "trust_level": 11072 }1073 }1074 ]1075 },1076 {1077 "fancy_title": "Can someone explain the benefits of Batches?",1078 "id": 212699,1079 "title": "Can someone explain the benefits of Batches?",1080 "slug": "can-someone-explain-the-benefits-of-batches",1081 "posts_count": 3,1082 "reply_count": 1,1083 "highest_post_number": 3,1084 "image_url": null,1085 "created_at": "2024-11-08T10:12:40.115Z",1086 "last_posted_at": "2024-11-08T15:24:42.721Z",1087 "bumped": true,1088 "bumped_at": "2024-11-08T15:24:42.721Z",1089 "archetype": "regular",1090 "unseen": false,1091 "pinned": false,1092 "unpinned": null,1093 "visible": true,1094 "closed": false,1095 "archived": false,1096 "bookmarked": null,1097 "liked": null,1098 "tags_descriptions": {},1099 "like_count": 2,1100 "views": 288,1101 "category_id": 8,1102 "featured_link": null,1103 "has_accepted_answer": false,1104 "posters": [1105 {1106 "extras": "latest",1107 "description": "Original Poster, Most Recent Poster",1108 "user": {1109 "id": 80540,1110 "username": "na50r",1111 "name": "",1112 "avatar_template": "/user_avatar/discuss.pytorch.org/na50r/{size}/73632_2.png",1113 "trust_level": 11114 }1115 },1116 {1117 "extras": null,1118 "description": "Frequent Poster",1119 "user": {1120 "id": 77701,1121 "username": "MLangner",1122 "name": "",1123 "avatar_template": "/letter_avatar_proxy/v4/letter/m/34f0e0/{size}.png",1124 "trust_level": 11125 }1126 }1127 ]1128 },1129 {1130 "fancy_title": "RuntimeError: The size of tensor a (2) must match the size of tensor b (0) at non-singleton dimension 1",1131 "id": 223491,1132 "title": "RuntimeError: The size of tensor a (2) must match the size of tensor b (0) at non-singleton dimension 1",1133 "slug": "runtimeerror-the-size-of-tensor-a-2-must-match-the-size-of-tensor-b-0-at-non-singleton-dimension-1",1134 "posts_count": 1,1135 "reply_count": 0,1136 "highest_post_number": 1,1137 "image_url": null,1138 "created_at": "2025-10-06T20:34:56.792Z",1139 "last_posted_at": "2025-10-06T20:34:56.850Z",1140 "bumped": true,1141 "bumped_at": "2025-10-06T20:34:56.850Z",1142 "archetype": "regular",1143 "unseen": false,1144 "pinned": false,1145 "unpinned": null,1146 "visible": true,1147 "closed": false,1148 "archived": false,1149 "bookmarked": null,1150 "liked": null,1151 "tags_descriptions": {},1152 "like_count": 0,1153 "views": 21,1154 "category_id": 8,1155 "featured_link": null,1156 "has_accepted_answer": false,1157 "posters": [1158 {1159 "extras": "latest single",1160 "description": "Original Poster, Most Recent Poster",1161 "user": {1162 "id": 78715,1163 "username": "pryce",1164 "name": "Pryce Houck",1165 "avatar_template": "/letter_avatar_proxy/v4/letter/p/5e9695/{size}.png",1166 "trust_level": 01167 }1168 }1169 ]1170 }1171 ],1172 "tags_descriptions": {},1173 "fancy_title": "A question for batch-training RNN",1174 "id": 222402,1175 "title": "A question for batch-training RNN",1176 "posts_count": 4,1177 "created_at": "2025-08-16T15:29:15.431Z",1178 "views": 40,1179 "reply_count": 2,1180 "like_count": 1,1181 "last_posted_at": "2025-08-19T02:50:54.214Z",1182 "visible": true,1183 "closed": false,1184 "archived": false,1185 "has_summary": false,1186 "archetype": "regular",1187 "slug": "a-question-for-batch-training-rnn",1188 "category_id": 8,1189 "word_count": 275,1190 "deleted_at": null,1191 "user_id": 74088,1192 "featured_link": null,1193 "pinned_globally": false,1194 "pinned_at": null,1195 "pinned_until": null,1196 "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/4/4/440da3b6c9137829f920dbb36869068895b48257.png",1197 "slow_mode_seconds": 0,1198 "draft": null,1199 "draft_key": "topic_222402",1200 "draft_sequence": null,