Anurag1734/cuda-error-resolution-analysis
07
1[2 {3 "post_stream": {4 "posts": [5 {6 "id": 332986,7 "name": "Nico Zhou",8 "username": "nicozhou",9 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png",10 "created_at": "2022-02-24T02:30:36.440Z",11 "cooked": "<p>I am doing a regression problem, and right now facing a problem where my model can only be trained with small amount of data, but when I increase the amount of data, the network does not learn(loss does not decreasing).</p>\n<p>Does anyone ever meet the same problem as me?</p>",12 "post_number": 1,13 "post_type": 1,14 "posts_count": 6,15 "updated_at": "2022-02-24T02:30:36.440Z",16 "reply_count": 0,17 "reply_to_post_number": null,18 "quote_count": 0,19 "incoming_link_count": 43,20 "reads": 8,21 "readers_count": 7,22 "score": 216.6,23 "yours": false,24 "topic_id": 144884,25 "topic_slug": "model-only-fits-for-small-dataset",26 "display_username": "Nico Zhou",27 "primary_group_name": null,28 "flair_name": null,29 "flair_url": null,30 "flair_bg_color": null,31 "flair_color": null,32 "flair_group_id": null,33 "badges_granted": [],34 "version": 1,35 "can_edit": false,36 "can_delete": false,37 "can_recover": false,38 "can_see_hidden_post": false,39 "can_wiki": false,40 "read": true,41 "user_title": null,42 "bookmarked": false,43 "actions_summary": [],44 "moderator": false,45 "admin": false,46 "staff": false,47 "user_id": 44912,48 "hidden": false,49 "trust_level": 1,50 "deleted_at": null,51 "user_deleted": false,52 "edit_reason": null,53 "can_view_edit_history": true,54 "wiki": false,55 "post_url": "/t/model-only-fits-for-small-dataset/144884/1",56 "can_accept_answer": false,57 "can_unaccept_answer": false,58 "accepted_answer": false,59 "topic_accepted_answer": null,60 "can_vote": false61 },62 {63 "id": 333001,64 "name": "Suho Cho",65 "username": "thecho7",66 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png",67 "created_at": "2022-02-24T04:58:26.185Z",68 "cooked": "<p>I don’t know how small the change of loss but we can typically observe that with a small model.<br>\nIt means your model is not enough big to fit the larger dataset.</p>\n<p>Scale up your model and show how it changes</p>",69 "post_number": 2,70 "post_type": 1,71 "posts_count": 6,72 "updated_at": "2022-02-24T04:58:26.185Z",73 "reply_count": 1,74 "reply_to_post_number": null,75 "quote_count": 0,76 "incoming_link_count": 1,77 "reads": 7,78 "readers_count": 6,79 "score": 11.4,80 "yours": false,81 "topic_id": 144884,82 "topic_slug": "model-only-fits-for-small-dataset",83 "display_username": "Suho Cho",84 "primary_group_name": null,85 "flair_name": null,86 "flair_url": null,87 "flair_bg_color": null,88 "flair_color": null,89 "flair_group_id": null,90 "badges_granted": [],91 "version": 1,92 "can_edit": false,93 "can_delete": false,94 "can_recover": false,95 "can_see_hidden_post": false,96 "can_wiki": false,97 "read": true,98 "user_title": "",99 "bookmarked": false,100 "actions_summary": [],101 "moderator": false,102 "admin": false,103 "staff": false,104 "user_id": 7263,105 "hidden": false,106 "trust_level": 2,107 "deleted_at": null,108 "user_deleted": false,109 "edit_reason": null,110 "can_view_edit_history": true,111 "wiki": false,112 "post_url": "/t/model-only-fits-for-small-dataset/144884/2",113 "can_accept_answer": false,114 "can_unaccept_answer": false,115 "accepted_answer": false,116 "topic_accepted_answer": null117 },118 {119 "id": 333034,120 "name": "Nico Zhou",121 "username": "nicozhou",122 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png",123 "created_at": "2022-02-24T06:56:37.759Z",124 "cooked": "<p>the loss I used is L1 Loss, and it started from around 0.4 and can only decrease to 0.3</p>",125 "post_number": 3,126 "post_type": 1,127 "posts_count": 6,128 "updated_at": "2022-02-24T06:56:37.759Z",129 "reply_count": 1,130 "reply_to_post_number": 2,131 "quote_count": 0,132 "incoming_link_count": 1,133 "reads": 5,134 "readers_count": 4,135 "score": 11.0,136 "yours": false,137 "topic_id": 144884,138 "topic_slug": "model-only-fits-for-small-dataset",139 "display_username": "Nico Zhou",140 "primary_group_name": null,141 "flair_name": null,142 "flair_url": null,143 "flair_bg_color": null,144 "flair_color": null,145 "flair_group_id": null,146 "badges_granted": [],147 "version": 1,148 "can_edit": false,149 "can_delete": false,150 "can_recover": false,151 "can_see_hidden_post": false,152 "can_wiki": false,153 "read": true,154 "user_title": null,155 "reply_to_user": {156 "id": 7263,157 "username": "thecho7",158 "name": "Suho Cho",159 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png"160 },161 "bookmarked": false,162 "actions_summary": [],163 "moderator": false,164 "admin": false,165 "staff": false,166 "user_id": 44912,167 "hidden": false,168 "trust_level": 1,169 "deleted_at": null,170 "user_deleted": false,171 "edit_reason": null,172 "can_view_edit_history": true,173 "wiki": false,174 "post_url": "/t/model-only-fits-for-small-dataset/144884/3",175 "can_accept_answer": false,176 "can_unaccept_answer": false,177 "accepted_answer": false,178 "topic_accepted_answer": null179 },180 {181 "id": 333042,182 "name": "Suho Cho",183 "username": "thecho7",184 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png",185 "created_at": "2022-02-24T07:43:13.396Z",186 "cooked": "<p>How small the loss value cannot exactly explain how good your model is.<br>\nIf you use <code>nn.L1Loss</code>, check whether you are averaging the loss by the size of batch.</p>\n<p>Anyway, scale up your model first.</p>",187 "post_number": 4,188 "post_type": 1,189 "posts_count": 6,190 "updated_at": "2022-02-24T07:43:13.396Z",191 "reply_count": 1,192 "reply_to_post_number": 3,193 "quote_count": 0,194 "incoming_link_count": 1,195 "reads": 5,196 "readers_count": 4,197 "score": 11.0,198 "yours": false,199 "topic_id": 144884,200 "topic_slug": "model-only-fits-for-small-dataset",201 "display_username": "Suho Cho",202 "primary_group_name": null,203 "flair_name": null,204 "flair_url": null,205 "flair_bg_color": null,206 "flair_color": null,207 "flair_group_id": null,208 "badges_granted": [],209 "version": 1,210 "can_edit": false,211 "can_delete": false,212 "can_recover": false,213 "can_see_hidden_post": false,214 "can_wiki": false,215 "read": true,216 "user_title": "",217 "reply_to_user": {218 "id": 44912,219 "username": "nicozhou",220 "name": "Nico Zhou",221 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png"222 },223 "bookmarked": false,224 "actions_summary": [],225 "moderator": false,226 "admin": false,227 "staff": false,228 "user_id": 7263,229 "hidden": false,230 "trust_level": 2,231 "deleted_at": null,232 "user_deleted": false,233 "edit_reason": null,234 "can_view_edit_history": true,235 "wiki": false,236 "post_url": "/t/model-only-fits-for-small-dataset/144884/4",237 "can_accept_answer": false,238 "can_unaccept_answer": false,239 "accepted_answer": false,240 "topic_accepted_answer": null241 },242 {243 "id": 333047,244 "name": "Nico Zhou",245 "username": "nicozhou",246 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png",247 "created_at": "2022-02-24T07:47:43.928Z",248 "cooked": "<p>yeah, I mean it is supposed to be decreasing to 0.01. But it could not. By the way, scaling model is like adding more parameters?</p>",249 "post_number": 5,250 "post_type": 1,251 "posts_count": 6,252 "updated_at": "2022-02-24T07:47:43.928Z",253 "reply_count": 1,254 "reply_to_post_number": 4,255 "quote_count": 0,256 "incoming_link_count": 0,257 "reads": 5,258 "readers_count": 4,259 "score": 6.0,260 "yours": false,261 "topic_id": 144884,262 "topic_slug": "model-only-fits-for-small-dataset",263 "display_username": "Nico Zhou",264 "primary_group_name": null,265 "flair_name": null,266 "flair_url": null,267 "flair_bg_color": null,268 "flair_color": null,269 "flair_group_id": null,270 "badges_granted": [],271 "version": 1,272 "can_edit": false,273 "can_delete": false,274 "can_recover": false,275 "can_see_hidden_post": false,276 "can_wiki": false,277 "read": true,278 "user_title": null,279 "reply_to_user": {280 "id": 7263,281 "username": "thecho7",282 "name": "Suho Cho",283 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png"284 },285 "bookmarked": false,286 "actions_summary": [],287 "moderator": false,288 "admin": false,289 "staff": false,290 "user_id": 44912,291 "hidden": false,292 "trust_level": 1,293 "deleted_at": null,294 "user_deleted": false,295 "edit_reason": null,296 "can_view_edit_history": true,297 "wiki": false,298 "post_url": "/t/model-only-fits-for-small-dataset/144884/5",299 "can_accept_answer": false,300 "can_unaccept_answer": false,301 "accepted_answer": false,302 "topic_accepted_answer": null303 },304 {305 "id": 333050,306 "name": "Suho Cho",307 "username": "thecho7",308 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png",309 "created_at": "2022-02-24T07:50:04.588Z",310 "cooked": "<p>Yes, stacking more layers or somehow you already know</p>",311 "post_number": 6,312 "post_type": 1,313 "posts_count": 6,314 "updated_at": "2022-02-24T07:50:12.062Z",315 "reply_count": 0,316 "reply_to_post_number": 5,317 "quote_count": 0,318 "incoming_link_count": 1,319 "reads": 5,320 "readers_count": 4,321 "score": 21.0,322 "yours": false,323 "topic_id": 144884,324 "topic_slug": "model-only-fits-for-small-dataset",325 "display_username": "Suho Cho",326 "primary_group_name": null,327 "flair_name": null,328 "flair_url": null,329 "flair_bg_color": null,330 "flair_color": null,331 "flair_group_id": null,332 "badges_granted": [],333 "version": 1,334 "can_edit": false,335 "can_delete": false,336 "can_recover": false,337 "can_see_hidden_post": false,338 "can_wiki": false,339 "read": true,340 "user_title": "",341 "reply_to_user": {342 "id": 44912,343 "username": "nicozhou",344 "name": "Nico Zhou",345 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png"346 },347 "bookmarked": false,348 "actions_summary": [349 {350 "id": 2,351 "count": 1352 }353 ],354 "moderator": false,355 "admin": false,356 "staff": false,357 "user_id": 7263,358 "hidden": false,359 "trust_level": 2,360 "deleted_at": null,361 "user_deleted": false,362 "edit_reason": null,363 "can_view_edit_history": true,364 "wiki": false,365 "post_url": "/t/model-only-fits-for-small-dataset/144884/6",366 "can_accept_answer": false,367 "can_unaccept_answer": false,368 "accepted_answer": false,369 "topic_accepted_answer": null370 }371 ],372 "stream": [373 332986,374 333001,375 333034,376 333042,377 333047,378 333050379 ]380 },381 "timeline_lookup": [382 [383 1,384 1340385 ]386 ],387 "suggested_topics": [388 {389 "fancy_title": "Is my code correct for calculating score during validation?",390 "id": 213685,391 "title": "Is my code correct for calculating score during validation?",392 "slug": "is-my-code-correct-for-calculating-score-during-validation",393 "posts_count": 2,394 "reply_count": 0,395 "highest_post_number": 2,396 "image_url": null,397 "created_at": "2024-12-02T08:46:38.230Z",398 "last_posted_at": "2024-12-02T08:47:20.551Z",399 "bumped": true,400 "bumped_at": "2024-12-02T08:47:20.551Z",401 "archetype": "regular",402 "unseen": false,403 "pinned": false,404 "unpinned": null,405 "visible": true,406 "closed": false,407 "archived": false,408 "bookmarked": null,409 "liked": null,410 "tags_descriptions": {},411 "like_count": 0,412 "views": 34,413 "category_id": 1,414 "featured_link": null,415 "has_accepted_answer": false,416 "posters": [417 {418 "extras": "latest single",419 "description": "Original Poster, Most Recent Poster",420 "user": {421 "id": 75241,422 "username": "jawi289p",423 "name": "",424 "avatar_template": "/user_avatar/discuss.pytorch.org/jawi289p/{size}/61673_2.png",425 "trust_level": 1426 }427 }428 ]429 },430 {431 "fancy_title": "RNN: Is it advisable initialize hidden state in training only once",432 "id": 212588,433 "title": "RNN: Is it advisable initialize hidden state in training only once",434 "slug": "rnn-is-it-advisable-initialize-hidden-state-in-training-only-once",435 "posts_count": 2,436 "reply_count": 0,437 "highest_post_number": 2,438 "image_url": null,439 "created_at": "2024-11-06T04:23:13.376Z",440 "last_posted_at": "2024-11-06T13:29:34.687Z",441 "bumped": true,442 "bumped_at": "2024-11-06T13:29:34.687Z",443 "archetype": "regular",444 "unseen": false,445 "pinned": false,446 "unpinned": null,447 "visible": true,448 "closed": false,449 "archived": false,450 "bookmarked": null,451 "liked": null,452 "tags_descriptions": {},453 "like_count": 0,454 "views": 112,455 "category_id": 1,456 "featured_link": null,457 "has_accepted_answer": false,458 "posters": [459 {460 "extras": null,461 "description": "Original Poster",462 "user": {463 "id": 80718,464 "username": "Ras",465 "name": "",466 "avatar_template": "/letter_avatar_proxy/v4/letter/r/58956e/{size}.png",467 "trust_level": 1468 }469 },470 {471 "extras": "latest",472 "description": "Most Recent Poster",473 "user": {474 "id": 80720,475 "username": "Michael_Looney",476 "name": "Michael Looney",477 "avatar_template": "/user_avatar/discuss.pytorch.org/michael_looney/{size}/73798_2.png",478 "trust_level": 0479 }480 }481 ]482 },483 {484 "fancy_title": "Torch svd grad show all zero, when only use Vt[-1]",485 "id": 215307,486 "title": "Torch svd grad show all zero, when only use Vt[-1]",487 "slug": "torch-svd-grad-show-all-zero-when-only-use-vt-1",488 "posts_count": 2,489 "reply_count": 0,490 "highest_post_number": 2,491 "image_url": null,492 "created_at": "2025-01-13T04:25:23.822Z",493 "last_posted_at": "2025-01-14T19:59:03.630Z",494 "bumped": true,495 "bumped_at": "2025-01-14T19:59:03.630Z",496 "archetype": "regular",497 "unseen": false,498 "pinned": false,499 "unpinned": null,500 "visible": true,501 "closed": false,502 "archived": false,503 "bookmarked": null,504 "liked": null,505 "tags_descriptions": {},506 "like_count": 0,507 "views": 61,508 "category_id": 1,509 "featured_link": null,510 "has_accepted_answer": false,511 "posters": [512 {513 "extras": null,514 "description": "Original Poster",515 "user": {516 "id": 82045,517 "username": "yangtaodummt",518 "name": "Yangtaodummt",519 "avatar_template": "/user_avatar/discuss.pytorch.org/yangtaodummt/{size}/75071_2.png",520 "trust_level": 0521 }522 },523 {524 "extras": "latest",525 "description": "Most Recent Poster",526 "user": {527 "id": 18088,528 "username": "KFrank",529 "name": "K. Frank",530 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",531 "trust_level": 2532 }533 }534 ]535 },536 {537 "fancy_title": "Alternative for CUDAExtension for intel GPU(xpu)",538 "id": 217384,539 "title": "Alternative for CUDAExtension for intel GPU(xpu)",540 "slug": "alternative-for-cudaextension-for-intel-gpu-xpu",541 "posts_count": 1,542 "reply_count": 0,543 "highest_post_number": 1,544 "image_url": null,545 "created_at": "2025-03-03T15:36:33.733Z",546 "last_posted_at": "2025-03-03T15:36:33.771Z",547 "bumped": true,548 "bumped_at": "2025-03-03T15:36:33.771Z",549 "archetype": "regular",550 "unseen": false,551 "pinned": false,552 "unpinned": null,553 "visible": true,554 "closed": false,555 "archived": false,556 "bookmarked": null,557 "liked": null,558 "tags_descriptions": {},559 "like_count": 0,560 "views": 49,561 "category_id": 1,562 "featured_link": null,563 "has_accepted_answer": false,564 "posters": [565 {566 "extras": "latest single",567 "description": "Original Poster, Most Recent Poster",568 "user": {569 "id": 79942,570 "username": "yash3056",571 "name": "Yash3056",572 "avatar_template": "/user_avatar/discuss.pytorch.org/yash3056/{size}/73076_2.png",573 "trust_level": 0574 }575 }576 ]577 },578 {579 "fancy_title": "Does flex_attention still require a sequence length of 128 or not?",580 "id": 219846,581 "title": "Does flex_attention still require a sequence length of 128 or not?",582 "slug": "does-flex-attention-still-require-a-sequence-length-of-128-or-not",583 "posts_count": 2,584 "reply_count": 0,585 "highest_post_number": 2,586 "image_url": null,587 "created_at": "2025-05-07T20:17:22.850Z",588 "last_posted_at": "2025-05-07T21:34:01.652Z",589 "bumped": true,590 "bumped_at": "2025-05-07T21:34:01.652Z",591 "archetype": "regular",592 "unseen": false,593 "pinned": false,594 "unpinned": null,595 "visible": true,596 "closed": false,597 "archived": false,598 "bookmarked": null,599 "liked": null,600 "tags_descriptions": {},601 "like_count": 1,602 "views": 91,603 "category_id": 1,604 "featured_link": null,605 "has_accepted_answer": true,606 "posters": [607 {608 "extras": null,609 "description": "Original Poster",610 "user": {611 "id": 970,612 "username": "divinho",613 "name": "",614 "avatar_template": "/letter_avatar_proxy/v4/letter/d/9dc877/{size}.png",615 "trust_level": 2616 }617 },618 {619 "extras": "latest",620 "description": "Most Recent Poster, Accepted Answer",621 "user": {622 "id": 42875,623 "username": "Chillee",624 "name": "Horace He",625 "avatar_template": "/user_avatar/discuss.pytorch.org/chillee/{size}/35574_2.png",626 "trust_level": 2627 }628 }629 ]630 }631 ],632 "tags_descriptions": {},633 "fancy_title": "Model only fits for small dataset",634 "id": 144884,635 "title": "Model only fits for small dataset",636 "posts_count": 6,637 "created_at": "2022-02-24T02:30:36.357Z",638 "views": 420,639 "reply_count": 4,640 "like_count": 1,641 "last_posted_at": "2022-02-24T07:50:04.588Z",642 "visible": true,643 "closed": false,644 "archived": false,645 "has_summary": false,646 "archetype": "regular",647 "slug": "model-only-fits-for-small-dataset",648 "category_id": 1,649 "word_count": 184,650 "deleted_at": null,651 "user_id": 44912,652 "featured_link": null,653 "pinned_globally": false,654 "pinned_at": null,655 "pinned_until": null,656 "image_url": null,657 "slow_mode_seconds": 0,658 "draft": null,659 "draft_key": "topic_144884",660 "draft_sequence": null,661 "unpinned": null,662 "pinned": false,663 "current_post_number": 1,664 "highest_post_number": 6,665 "deleted_by": null,666 "actions_summary": [667 {668 "id": 4,669 "count": 0,670 "hidden": false,671 "can_act": false672 },673 {674 "id": 8,675 "count": 0,676 "hidden": false,677 "can_act": false678 },679 {680 "id": 10,681 "count": 0,682 "hidden": false,683 "can_act": false684 },685 {686 "id": 7,687 "count": 0,688 "hidden": false,689 "can_act": false690 }691 ],692 "chunk_size": 20,693 "bookmarked": false,694 "topic_timer": null,695 "message_bus_last_id": 0,696 "participant_count": 2,697 "show_read_indicator": false,698 "thumbnails": null,699 "slow_mode_enabled_until": null,700 "can_vote": false,701 "vote_count": 0,702 "user_voted": false,703 "discourse_zendesk_plugin_zendesk_id": null,704 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",705 "details": {706 "can_edit": false,707 "notification_level": 1,708 "participants": [709 {710 "id": 7263,711 "username": "thecho7",712 "name": "Suho Cho",713 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png",714 "post_count": 3,715 "primary_group_name": null,716 "flair_name": null,717 "flair_url": null,718 "flair_color": null,719 "flair_bg_color": null,720 "flair_group_id": null,721 "trust_level": 2722 },723 {724 "id": 44912,725 "username": "nicozhou",726 "name": "Nico Zhou",727 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png",728 "post_count": 3,729 "primary_group_name": null,730 "flair_name": null,731 "flair_url": null,732 "flair_color": null,733 "flair_bg_color": null,734 "flair_group_id": null,735 "trust_level": 1736 }737 ],738 "created_by": {739 "id": 44912,740 "username": "nicozhou",741 "name": "Nico Zhou",742 "avatar_template": "/user_avatar/discuss.pytorch.org/nicozhou/{size}/37762_2.png"743 },744 "last_poster": {745 "id": 7263,746 "username": "thecho7",747 "name": "Suho Cho",748 "avatar_template": "/letter_avatar_proxy/v4/letter/t/eada6e/{size}.png"749 }750 },751 "bookmarks": []752 },753 {754 "post_stream": {755 "posts": [756 {757 "id": 333037,758 "name": "Andreas Habring",759 "username": "AndreasH",760 "avatar_template": "/user_avatar/discuss.pytorch.org/andreash/{size}/44707_2.png",761 "created_at": "2022-02-24T07:29:32.786Z",762 "cooked": "<p>Hey everybody,</p>\n<p>I have a question.<br>\nI wanted to understand what exactly is done within batch normalization. In the documentation it is stated, that BatchNorm1d performs batch normalization as stated in this paper</p><aside class=\"onebox pdf\" data-onebox-src=\"https://arxiv.org/pdf/1502.03167.pdf\">\n <header class=\"source\">\n\n <a href=\"https://arxiv.org/pdf/1502.03167.pdf\" target=\"_blank\" rel=\"noopener nofollow ugc\">arxiv.org</a>\n </header>\n\n <article class=\"onebox-body\">\n <a href=\"https://arxiv.org/pdf/1502.03167.pdf\" target=\"_blank\" rel=\"noopener nofollow ugc\"><span class=\"pdf-onebox-logo\"></span></a>\n\n<h3><a href=\"https://arxiv.org/pdf/1502.03167.pdf\" target=\"_blank\" rel=\"noopener nofollow ugc\">1502.03167.pdf</a></h3>\n\n <p class=\"filesize\">169.48 KB</p>\n\n </article>\n\n <div class=\"onebox-metadata\">\n \n \n </div>\n\n <div style=\"clear: both\"></div>\n</aside>\n<p>\nI do, however, think that this is not true.<br>\nThis is what BatchNorm1d does (using default parameters, I tested and it results in the correct output):</p>\n<p>Input: Tensor X with shape (N,C,L)</p>\n<p>mean_X = torch.mean(X, dim=(0,2), keepdims=True)<br>\nvar_X = torch.mean((X-mean_X)*(x-mean_X),dim=(0,2), keepdims=True)<br>\nBN = (x-mean_x)/np.sqrt(var_x+1e-5)</p>\n<p>Output: BN</p>\n<p>So the mean and variance are computed over all input axes (accept for C which allows for different channels). In the paper, however, they state that mean and variance are computed over the batch size, which I think corresponds to setting dim=0 above.<br>\nCan someone confirm or explain.</p>",763 "post_number": 1,764 "post_type": 1,765 "posts_count": 1,766 "updated_at": "2022-02-24T07:30:20.912Z",767 "reply_count": 0,768 "reply_to_post_number": null,769 "quote_count": 0,770 "incoming_link_count": 12,771 "reads": 5,772 "readers_count": 4,773 "score": 61.0,774 "yours": false,775 "topic_id": 144901,776 "topic_slug": "batch-normalization-does-not-what-it-states",777 "display_username": "Andreas Habring",778 "primary_group_name": null,779 "flair_name": null,780 "flair_url": null,781 "flair_bg_color": null,782 "flair_color": null,783 "flair_group_id": null,784 "badges_granted": [],785 "version": 1,786 "can_edit": false,787 "can_delete": false,788 "can_recover": false,789 "can_see_hidden_post": false,790 "can_wiki": false,791 "link_counts": [792 {793 "url": "https://arxiv.org/pdf/1502.03167.pdf",794 "internal": false,795 "reflection": false,796 "clicks": 0797 }798 ],799 "read": true,800 "user_title": null,801 "bookmarked": false,802 "actions_summary": [],803 "moderator": false,804 "admin": false,805 "staff": false,806 "user_id": 46933,807 "hidden": false,808 "trust_level": 1,809 "deleted_at": null,810 "user_deleted": false,811 "edit_reason": null,812 "can_view_edit_history": true,813 "wiki": false,814 "post_url": "/t/batch-normalization-does-not-what-it-states/144901/1",815 "can_accept_answer": false,816 "can_unaccept_answer": false,817 "accepted_answer": false,818 "topic_accepted_answer": null,819 "can_vote": false820 }821 ],822 "stream": [823 333037824 ]825 },826 "timeline_lookup": [827 [828 1,829 1340830 ]831 ],832 "suggested_topics": [833 {834 "fancy_title": "Training accuracy significantly decreases and doesn’t go back up when loading from a checkpoint",835 "id": 217328,836 "title": "Training accuracy significantly decreases and doesn't go back up when loading from a checkpoint",837 "slug": "training-accuracy-significantly-decreases-and-doesnt-go-back-up-when-loading-from-a-checkpoint",838 "posts_count": 3,839 "reply_count": 2,840 "highest_post_number": 4,841 "image_url": null,842 "created_at": "2025-03-02T00:10:21.117Z",843 "last_posted_at": "2025-03-02T21:49:49.170Z",844 "bumped": true,845 "bumped_at": "2025-03-02T21:49:49.170Z",846 "archetype": "regular",847 "unseen": false,848 "pinned": false,849 "unpinned": null,850 "visible": true,851 "closed": false,852 "archived": false,853 "bookmarked": null,854 "liked": null,855 "tags_descriptions": {},856 "like_count": 0,857 "views": 41,858 "category_id": 1,859 "featured_link": null,860 "has_accepted_answer": false,861 "posters": [862 {863 "extras": "latest",864 "description": "Original Poster, Most Recent Poster",865 "user": {866 "id": 83008,867 "username": "EaswarGn",868 "name": "Easwar Gn",869 "avatar_template": "/user_avatar/discuss.pytorch.org/easwargn/{size}/75936_2.png",870 "trust_level": 1871 }872 },873 {874 "extras": null,875 "description": "Frequent Poster",876 "user": {877 "id": 3534,878 "username": "ptrblck",879 "name": "",880 "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",881 "admin": true,882 "moderator": true,883 "trust_level": 2884 }885 }886 ]887 },888 {889 "fancy_title": "Which activation function for multi-class classification gives true probability?",890 "id": 213065,891 "title": "Which activation function for multi-class classification gives true probability?",892 "slug": "which-activation-function-for-multi-class-classification-gives-true-probability",893 "posts_count": 2,894 "reply_count": 0,895 "highest_post_number": 2,896 "image_url": null,897 "created_at": "2024-11-17T07:26:52.131Z",898 "last_posted_at": "2024-11-17T21:03:13.613Z",899 "bumped": true,900 "bumped_at": "2024-11-18T08:58:42.971Z",901 "archetype": "regular",902 "unseen": false,903 "pinned": false,904 "unpinned": null,905 "visible": true,906 "closed": false,907 "archived": false,908 "bookmarked": null,909 "liked": null,910 "tags_descriptions": {},911 "like_count": 1,912 "views": 117,913 "category_id": 1,914 "featured_link": null,915 "has_accepted_answer": true,916 "posters": [917 {918 "extras": null,919 "description": "Original Poster",920 "user": {921 "id": 50872,922 "username": "laro",923 "name": "amit",924 "avatar_template": "/user_avatar/discuss.pytorch.org/laro/{size}/47125_2.png",925 "trust_level": 1926 }927 },928 {929 "extras": "latest",930 "description": "Most Recent Poster, Accepted Answer",931 "user": {932 "id": 18088,933 "username": "KFrank",934 "name": "K. Frank",935 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",936 "trust_level": 2937 }938 }939 ]940 },941 {942 "fancy_title": "Find maximum length of consecutive zeros in each row",943 "id": 213871,944 "title": "Find maximum length of consecutive zeros in each row",945 "slug": "find-maximum-length-of-consecutive-zeros-in-each-row",946 "posts_count": 3,947 "reply_count": 0,948 "highest_post_number": 3,949 "image_url": null,950 "created_at": "2024-12-05T16:46:38.153Z",951 "last_posted_at": "2024-12-08T19:58:20.444Z",952 "bumped": true,953 "bumped_at": "2024-12-08T19:58:20.444Z",954 "archetype": "regular",955 "unseen": false,956 "pinned": false,957 "unpinned": null,958 "visible": true,959 "closed": false,960 "archived": false,961 "bookmarked": null,962 "liked": null,963 "tags_descriptions": {},964 "like_count": 0,965 "views": 215,966 "category_id": 1,967 "featured_link": null,968 "has_accepted_answer": false,969 "posters": [970 {971 "extras": null,972 "description": "Original Poster",973 "user": {974 "id": 81339,975 "username": "Paulo_Nascimento",976 "name": "Paulo Nascimento",977 "avatar_template": "/user_avatar/discuss.pytorch.org/paulo_nascimento/{size}/72888_2.png",978 "trust_level": 0979 }980 },981 {982 "extras": null,983 "description": "Frequent Poster",984 "user": {985 "id": 72430,986 "username": "Eduardo_Lawson",987 "name": "Eduardo Lawson da Silva",988 "avatar_template": "/user_avatar/discuss.pytorch.org/eduardo_lawson/{size}/66899_2.png",989 "trust_level": 2990 }991 },992 {993 "extras": "latest",994 "description": "Most Recent Poster",995 "user": {996 "id": 18088,997 "username": "KFrank",998 "name": "K. Frank",999 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",1000 "trust_level": 21001 }1002 }1003 ]1004 },1005 {1006 "fancy_title": "Why is my generator model M_gen not training when optimizing based on the classifier model M_cls?",1007 "id": 216263,1008 "title": "Why is my generator model M_gen not training when optimizing based on the classifier model M_cls?",1009 "slug": "why-is-my-generator-model-m-gen-not-training-when-optimizing-based-on-the-classifier-model-m-cls",1010 "posts_count": 7,1011 "reply_count": 4,1012 "highest_post_number": 7,1013 "image_url": null,1014 "created_at": "2025-02-05T10:43:44.579Z",1015 "last_posted_at": "2025-02-09T22:42:53.438Z",1016 "bumped": true,1017 "bumped_at": "2025-02-09T22:42:53.438Z",1018 "archetype": "regular",1019 "unseen": false,1020 "pinned": false,1021 "unpinned": null,1022 "visible": true,1023 "closed": false,1024 "archived": false,1025 "bookmarked": null,1026 "liked": null,1027 "tags_descriptions": {},1028 "like_count": 0,1029 "views": 119,1030 "category_id": 1,1031 "featured_link": null,1032 "has_accepted_answer": false,1033 "posters": [1034 {1035 "extras": "latest",1036 "description": "Original Poster, Most Recent Poster",1037 "user": {1038 "id": 82500,1039 "username": "Brayn_O_Conner",1040 "name": "Brayn O'Conner",1041 "avatar_template": "/user_avatar/discuss.pytorch.org/brayn_o_conner/{size}/75477_2.png",1042 "trust_level": 01043 }1044 },1045 {1046 "extras": null,1047 "description": "Frequent Poster",1048 "user": {1049 "id": 18088,1050 "username": "KFrank",1051 "name": "K. Frank",1052 "avatar_template": "/letter_avatar_proxy/v4/letter/k/ecb155/{size}.png",1053 "trust_level": 21054 }1055 }1056 ]1057 },1058 {1059 "fancy_title": "Unusual Behavior in Validation Batches - 3D DenseNet121 model",1060 "id": 217716,1061 "title": "Unusual Behavior in Validation Batches - 3D DenseNet121 model",1062 "slug": "unusual-behavior-in-validation-batches-3d-densenet121-model",1063 "posts_count": 1,1064 "reply_count": 0,1065 "highest_post_number": 1,1066 "image_url": null,1067 "created_at": "2025-03-11T23:50:02.669Z",1068 "last_posted_at": "2025-03-11T23:50:02.715Z",1069 "bumped": true,1070 "bumped_at": "2025-03-11T23:50:02.715Z",1071 "archetype": "regular",1072 "unseen": false,1073 "pinned": false,1074 "unpinned": null,1075 "visible": true,1076 "closed": false,1077 "archived": false,1078 "bookmarked": null,1079 "liked": null,1080 "tags_descriptions": {},1081 "like_count": 0,1082 "views": 37,1083 "category_id": 1,1084 "featured_link": null,1085 "has_accepted_answer": false,1086 "posters": [1087 {1088 "extras": "latest single",1089 "description": "Original Poster, Most Recent Poster",1090 "user": {1091 "id": 83202,1092 "username": "anaf_sousa",1093 "name": "Ana Margarida Sousa",1094 "avatar_template": "/letter_avatar_proxy/v4/letter/a/85e7bf/{size}.png",1095 "trust_level": 11096 }1097 }1098 ]1099 }1100 ],1101 "tags_descriptions": {},1102 "fancy_title": "Batch normalization does not what it states",1103 "id": 144901,1104 "title": "Batch normalization does not what it states",1105 "posts_count": 1,1106 "created_at": "2022-02-24T07:29:32.721Z",1107 "views": 345,1108 "reply_count": 0,1109 "like_count": 0,1110 "last_posted_at": "2022-02-24T07:29:32.786Z",1111 "visible": true,1112 "closed": false,1113 "archived": false,1114 "has_summary": false,1115 "archetype": "regular",1116 "slug": "batch-normalization-does-not-what-it-states",1117 "category_id": 1,1118 "word_count": 154,1119 "deleted_at": null,1120 "user_id": 46933,1121 "featured_link": null,1122 "pinned_globally": false,1123 "pinned_at": null,1124 "pinned_until": null,1125 "image_url": null,1126 "slow_mode_seconds": 0,1127 "draft": null,1128 "draft_key": "topic_144901",1129 "draft_sequence": null,1130 "unpinned": null,1131 "pinned": false,1132 "current_post_number": 1,1133 "highest_post_number": 1,1134 "deleted_by": null,1135 "actions_summary": [1136 {1137 "id": 4,1138 "count": 0,1139 "hidden": false,1140 "can_act": false1141 },1142 {1143 "id": 8,1144 "count": 0,1145 "hidden": false,1146 "can_act": false1147 },1148 {1149 "id": 10,1150 "count": 0,1151 "hidden": false,1152 "can_act": false1153 },1154 {1155 "id": 7,1156 "count": 0,1157 "hidden": false,1158 "can_act": false1159 }1160 ],1161 "chunk_size": 20,1162 "bookmarked": false,1163 "topic_timer": null,1164 "message_bus_last_id": 0,1165 "participant_count": 1,1166 "show_read_indicator": false,1167 "thumbnails": null,1168 "slow_mode_enabled_until": null,1169 "can_vote": false,1170 "vote_count": 0,1171 "user_voted": false,1172 "discourse_zendesk_plugin_zendesk_id": null,1173 "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",1174 "details": {1175 "can_edit": false,1176 "notification_level": 1,1177 "participants": [1178 {1179 "id": 46933,1180 "username": "AndreasH",1181 "name": "Andreas Habring",1182 "avatar_template": "/user_avatar/discuss.pytorch.org/andreash/{size}/44707_2.png",1183 "post_count": 1,1184 "primary_group_name": null,1185 "flair_name": null,1186 "flair_url": null,1187 "flair_color": null,1188 "flair_bg_color": null,1189 "flair_group_id": null,1190 "trust_level": 11191 }1192 ],1193 "created_by": {1194 "id": 46933,1195 "username": "AndreasH",1196 "name": "Andreas Habring",1197 "avatar_template": "/user_avatar/discuss.pytorch.org/andreash/{size}/44707_2.png"1198 },1199 "last_poster": {1200 "id": 46933,