CoolFace
Datasetpublic

Anurag1734/cuda-error-resolution-analysis

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes7downloads
topics_batch_427.json65575 linesDownload Raw Back to raw
1[2  {3    "post_stream": {4      "posts": [5        {6          "id": 232380,7          "name": "翔 王",8          "username": "wangxiang2713",9          "avatar_template": "/user_avatar/discuss.pytorch.org/wangxiang2713/{size}/29236_2.png",10          "created_at": "2020-09-29T07:26:46.722Z",11          "cooked": "<p>Hi, i want to detect Bert model structure in pytorch model and convert the structure by Nvidia faster transformer op automatically.<br>\nIs there any existing project? If not, i want to develop one, so should i develop on origin pytorch or TorchScript? Should i develop a pass to detect Bert in TorchScript IR and replace it by faster transformer op?<br>\nThankyou very much!</p>",12          "post_number": 1,13          "post_type": 1,14          "posts_count": 1,15          "updated_at": "2020-09-29T07:26:46.722Z",16          "reply_count": 0,17          "reply_to_post_number": null,18          "quote_count": 0,19          "incoming_link_count": 105,20          "reads": 6,21          "readers_count": 5,22          "score": 526.2,23          "yours": false,24          "topic_id": 97700,25          "topic_slug": "how-to-convert-pytorch-model-to-nvidia-faster-transformer",26          "display_username": "翔 王",27          "primary_group_name": null,28          "flair_name": null,29          "flair_url": null,30          "flair_bg_color": null,31          "flair_color": null,32          "flair_group_id": null,33          "badges_granted": [],34          "version": 1,35          "can_edit": false,36          "can_delete": false,37          "can_recover": false,38          "can_see_hidden_post": false,39          "can_wiki": false,40          "read": true,41          "user_title": null,42          "bookmarked": false,43          "actions_summary": [],44          "moderator": false,45          "admin": false,46          "staff": false,47          "user_id": 37135,48          "hidden": false,49          "trust_level": 1,50          "deleted_at": null,51          "user_deleted": false,52          "edit_reason": null,53          "can_view_edit_history": true,54          "wiki": false,55          "post_url": "/t/how-to-convert-pytorch-model-to-nvidia-faster-transformer/97700/1",56          "can_accept_answer": false,57          "can_unaccept_answer": false,58          "accepted_answer": false,59          "topic_accepted_answer": null,60          "can_vote": false61        }62      ],63      "stream": [64        23238065      ]66    },67    "timeline_lookup": [68      [69        1,70        185371      ]72    ],73    "suggested_topics": [74      {75        "fancy_title": "Importing torchtext",76        "id": 216012,77        "title": "Importing torchtext",78        "slug": "importing-torchtext",79        "posts_count": 2,80        "reply_count": 0,81        "highest_post_number": 2,82        "image_url": null,83        "created_at": "2025-01-29T05:38:25.379Z",84        "last_posted_at": "2025-02-03T00:43:43.359Z",85        "bumped": true,86        "bumped_at": "2025-02-03T00:43:43.359Z",87        "archetype": "regular",88        "unseen": false,89        "pinned": false,90        "unpinned": null,91        "visible": true,92        "closed": false,93        "archived": false,94        "bookmarked": null,95        "liked": null,96        "tags_descriptions": {},97        "like_count": 1,98        "views": 378,99        "category_id": 8,100        "featured_link": null,101        "has_accepted_answer": false,102        "posters": [103          {104            "extras": null,105            "description": "Original Poster",106            "user": {107              "id": 81384,108              "username": "Dristro",109              "name": null,110              "avatar_template": "/user_avatar/discuss.pytorch.org/dristro/{size}/75367_2.png",111              "trust_level": 1112            }113          },114          {115            "extras": "latest",116            "description": "Most Recent Poster",117            "user": {118              "id": 1438,119              "username": "vdw",120              "name": "Chris",121              "avatar_template": "/user_avatar/discuss.pytorch.org/vdw/{size}/10074_2.png",122              "trust_level": 2123            }124          }125        ]126      },127      {128        "fancy_title": "How does one set the pad token correctly (not to eos) during fine-tuning to avoid model not predicting EOS?",129        "id": 213619,130        "title": "How does one set the pad token correctly (not to eos) during fine-tuning to avoid model not predicting EOS?",131        "slug": "how-does-one-set-the-pad-token-correctly-not-to-eos-during-fine-tuning-to-avoid-model-not-predicting-eos",132        "posts_count": 1,133        "reply_count": 0,134        "highest_post_number": 1,135        "image_url": null,136        "created_at": "2024-11-29T17:47:47.630Z",137        "last_posted_at": "2024-11-29T17:47:47.686Z",138        "bumped": true,139        "bumped_at": "2024-11-29T17:47:47.686Z",140        "archetype": "regular",141        "unseen": false,142        "pinned": false,143        "unpinned": null,144        "visible": true,145        "closed": false,146        "archived": false,147        "bookmarked": null,148        "liked": null,149        "tags_descriptions": {},150        "like_count": 0,151        "views": 1062,152        "category_id": 8,153        "featured_link": null,154        "has_accepted_answer": false,155        "posters": [156          {157            "extras": "latest single",158            "description": "Original Poster, Most Recent Poster",159            "user": {160              "id": 2282,161              "username": "Brando_Miranda",162              "name": "MirandaAgent",163              "avatar_template": "/user_avatar/discuss.pytorch.org/brando_miranda/{size}/14355_2.png",164              "trust_level": 2165            }166          }167        ]168      },169      {170        "fancy_title": "Training starting again in sampling code",171        "id": 212724,172        "title": "Training starting again in sampling code",173        "slug": "training-starting-again-in-sampling-code",174        "posts_count": 4,175        "reply_count": 2,176        "highest_post_number": 4,177        "image_url": null,178        "created_at": "2024-11-09T04:13:37.209Z",179        "last_posted_at": "2024-11-09T17:07:36.146Z",180        "bumped": true,181        "bumped_at": "2024-11-09T17:07:36.146Z",182        "archetype": "regular",183        "unseen": false,184        "pinned": false,185        "unpinned": null,186        "visible": true,187        "closed": false,188        "archived": false,189        "bookmarked": null,190        "liked": null,191        "tags_descriptions": {},192        "like_count": 2,193        "views": 63,194        "category_id": 8,195        "featured_link": null,196        "has_accepted_answer": true,197        "posters": [198          {199            "extras": null,200            "description": "Original Poster",201            "user": {202              "id": 80780,203              "username": "Rajatavaa",204              "name": "Rajatava Ghosh",205              "avatar_template": "/user_avatar/discuss.pytorch.org/rajatavaa/{size}/73875_2.png",206              "trust_level": 0207            }208          },209          {210            "extras": "latest",211            "description": "Most Recent Poster, Accepted Answer",212            "user": {213              "id": 3534,214              "username": "ptrblck",215              "name": "",216              "avatar_template": "/user_avatar/discuss.pytorch.org/ptrblck/{size}/1823_2.png",217              "admin": true,218              "moderator": true,219              "trust_level": 2220            }221          }222        ]223      },224      {225        "fancy_title": "Building a Model for Multi-Output Embedding Generation: Seeking Advice and Insights",226        "id": 214969,227        "title": "Building a Model for Multi-Output Embedding Generation: Seeking Advice and Insights",228        "slug": "building-a-model-for-multi-output-embedding-generation-seeking-advice-and-insights",229        "posts_count": 1,230        "reply_count": 0,231        "highest_post_number": 1,232        "image_url": null,233        "created_at": "2025-01-04T15:22:48.969Z",234        "last_posted_at": "2025-01-04T15:22:49.013Z",235        "bumped": true,236        "bumped_at": "2025-01-04T16:41:22.898Z",237        "archetype": "regular",238        "unseen": false,239        "pinned": false,240        "unpinned": null,241        "visible": true,242        "closed": false,243        "archived": false,244        "bookmarked": null,245        "liked": null,246        "tags_descriptions": {},247        "like_count": 0,248        "views": 53,249        "category_id": 8,250        "featured_link": null,251        "has_accepted_answer": false,252        "posters": [253          {254            "extras": "latest single",255            "description": "Original Poster, Most Recent Poster",256            "user": {257              "id": 81874,258              "username": "eyalbenbarouch1997",259              "name": "DataSnake",260              "avatar_template": "/user_avatar/discuss.pytorch.org/eyalbenbarouch1997/{size}/74899_2.png",261              "trust_level": 0262            }263          }264        ]265      },266      {267        "fancy_title": "A Simple LSTM stuck into label flipping",268        "id": 220004,269        "title": "A Simple LSTM stuck into label flipping",270        "slug": "a-simple-lstm-stuck-into-label-flipping",271        "posts_count": 5,272        "reply_count": 3,273        "highest_post_number": 5,274        "image_url": null,275        "created_at": "2025-05-13T19:44:43.099Z",276        "last_posted_at": "2025-05-15T08:28:26.193Z",277        "bumped": true,278        "bumped_at": "2025-05-15T08:28:26.193Z",279        "archetype": "regular",280        "unseen": false,281        "pinned": false,282        "unpinned": null,283        "visible": true,284        "closed": false,285        "archived": false,286        "bookmarked": null,287        "liked": null,288        "tags_descriptions": {},289        "like_count": 0,290        "views": 72,291        "category_id": 8,292        "featured_link": null,293        "has_accepted_answer": true,294        "posters": [295          {296            "extras": "latest",297            "description": "Original Poster, Most Recent Poster",298            "user": {299              "id": 84265,300              "username": "subhojitdas",301              "name": "Subhojit Das",302              "avatar_template": "/user_avatar/discuss.pytorch.org/subhojitdas/{size}/77001_2.png",303              "trust_level": 1304            }305          },306          {307            "extras": null,308            "description": "Frequent Poster, Accepted Answer",309            "user": {310              "id": 1438,311              "username": "vdw",312              "name": "Chris",313              "avatar_template": "/user_avatar/discuss.pytorch.org/vdw/{size}/10074_2.png",314              "trust_level": 2315            }316          }317        ]318      }319    ],320    "tags_descriptions": {},321    "fancy_title": "How to convert pytorch model to Nvidia faster transformer?",322    "id": 97700,323    "title": "How to convert pytorch model to Nvidia faster transformer?",324    "posts_count": 1,325    "created_at": "2020-09-29T07:26:46.660Z",326    "views": 435,327    "reply_count": 0,328    "like_count": 0,329    "last_posted_at": "2020-09-29T07:26:46.722Z",330    "visible": true,331    "closed": false,332    "archived": false,333    "has_summary": false,334    "archetype": "regular",335    "slug": "how-to-convert-pytorch-model-to-nvidia-faster-transformer",336    "category_id": 8,337    "word_count": 63,338    "deleted_at": null,339    "user_id": 37135,340    "featured_link": null,341    "pinned_globally": false,342    "pinned_at": null,343    "pinned_until": null,344    "image_url": null,345    "slow_mode_seconds": 0,346    "draft": null,347    "draft_key": "topic_97700",348    "draft_sequence": null,349    "unpinned": null,350    "pinned": false,351    "current_post_number": 1,352    "highest_post_number": 1,353    "deleted_by": null,354    "actions_summary": [355      {356        "id": 4,357        "count": 0,358        "hidden": false,359        "can_act": false360      },361      {362        "id": 8,363        "count": 0,364        "hidden": false,365        "can_act": false366      },367      {368        "id": 10,369        "count": 0,370        "hidden": false,371        "can_act": false372      },373      {374        "id": 7,375        "count": 0,376        "hidden": false,377        "can_act": false378      }379    ],380    "chunk_size": 20,381    "bookmarked": false,382    "topic_timer": null,383    "message_bus_last_id": 0,384    "participant_count": 1,385    "show_read_indicator": false,386    "thumbnails": null,387    "slow_mode_enabled_until": null,388    "can_vote": false,389    "vote_count": 0,390    "user_voted": false,391    "discourse_zendesk_plugin_zendesk_id": null,392    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",393    "details": {394      "can_edit": false,395      "notification_level": 1,396      "participants": [397        {398          "id": 37135,399          "username": "wangxiang2713",400          "name": "翔 王",401          "avatar_template": "/user_avatar/discuss.pytorch.org/wangxiang2713/{size}/29236_2.png",402          "post_count": 1,403          "primary_group_name": null,404          "flair_name": null,405          "flair_url": null,406          "flair_color": null,407          "flair_bg_color": null,408          "flair_group_id": null,409          "trust_level": 1410        }411      ],412      "created_by": {413        "id": 37135,414        "username": "wangxiang2713",415        "name": "翔 王",416        "avatar_template": "/user_avatar/discuss.pytorch.org/wangxiang2713/{size}/29236_2.png"417      },418      "last_poster": {419        "id": 37135,420        "username": "wangxiang2713",421        "name": "翔 王",422        "avatar_template": "/user_avatar/discuss.pytorch.org/wangxiang2713/{size}/29236_2.png"423      }424    },425    "bookmarks": []426  },427  {428    "post_stream": {429      "posts": [430        {431          "id": 232367,432          "name": "Hari Krishnan",433          "username": "Hari_Krishnan",434          "avatar_template": "/user_avatar/discuss.pytorch.org/hari_krishnan/{size}/16377_2.png",435          "created_at": "2020-09-29T06:10:59.893Z",436          "cooked": "<p>I trained a Transformer based Language Model using code available in PyTorch documentation, the results are worse than a RNN based LM. The code I’m using is from this part of documentation <a href=\"https://pytorch.org/tutorials/beginner/transformer_tutorial.html\" rel=\"noopener nofollow ugc\">https://pytorch.org/tutorials/beginner/transformer_tutorial.html</a>.</p>\n<p>Here are some output examples</p>\n<pre><code class=\"lang-auto\">&lt;eos&gt; @ settlement heavy of , , lined the she &lt;unk&gt; of . interception the dried . , would his \n= . , losses the and and the &lt;unk&gt; was , the the , &lt;unk&gt; &lt;unk&gt; the and be first \n\n= 1 was rains ireland and starting with hairy had found the &lt;unk&gt; to possibility heads other which receive gift \n= @ the , , the in the , been in &lt;unk&gt; , the of , , are the , \n\nvalkyria rebounds rapid over . truely their lead &lt;unk&gt; recently at year sharif take of as symptoms usually an for \nchronicles , and the the , first to , been the , , a the the , &lt;unk&gt; average the \n\nchronicles , in the since and drive were leaves left the , , a carpenter ear may happen index seizing \niii @ the storm the the , the the the &lt;unk&gt; and and touchdown , , be , , the \n\niii 3 particular island the son at built are the site he the 56 becoming ornaments include every @-@ on \n, @ by , republic , the in the &lt;unk&gt; , was &lt;unk&gt; @-@ the , &lt;unk&gt; year &lt;unk&gt; the \n\n= @ , , 1960s &lt;unk&gt; their during crowded &lt;unk&gt; . was cambridge – a . &lt;unk&gt; few linked telling \n= . the the , , first the , , the a , 0 &lt;unk&gt; the , years to the \n</code></pre>\n<p>You can see that text generated doesn’t even remotely make sense and is riddled with grammatical mistakes.</p>\n<p>Someone else has posted the same issue in the forums and is currently unanswered. Here’s the link for that <a href=\"https://discuss.pytorch.org/t/transformer-model-of-language-model-example-performs-worse-much/87448\" class=\"inline-onebox\">Transformer model of Language Model Example performs worse much</a></p>\n<p>Theoretically, wouldn’t transformers produce better results than RNN? What is the cause for this.</p>",437          "post_number": 1,438          "post_type": 1,439          "posts_count": 1,440          "updated_at": "2020-09-29T07:17:29.382Z",441          "reply_count": 0,442          "reply_to_post_number": null,443          "quote_count": 0,444          "incoming_link_count": 37,445          "reads": 8,446          "readers_count": 7,447          "score": 186.6,448          "yours": false,449          "topic_id": 97691,450          "topic_slug": "transformer-lm-in-pytorch-documentation-performs-poorly",451          "display_username": "Hari Krishnan",452          "primary_group_name": null,453          "flair_name": null,454          "flair_url": null,455          "flair_bg_color": null,456          "flair_color": null,457          "flair_group_id": null,458          "badges_granted": [],459          "version": 2,460          "can_edit": false,461          "can_delete": false,462          "can_recover": false,463          "can_see_hidden_post": false,464          "can_wiki": false,465          "link_counts": [466            {467              "url": "https://discuss.pytorch.org/t/transformer-model-of-language-model-example-performs-worse-much/87448",468              "internal": true,469              "reflection": false,470              "title": "Transformer model of Language Model Example performs worse much",471              "clicks": 8472            },473            {474              "url": "https://pytorch.org/tutorials/beginner/transformer_tutorial.html",475              "internal": false,476              "reflection": false,477              "title": "Sequence-to-Sequence Modeling with nn.Transformer and TorchText — PyTorch Tutorials 1.6.0 documentation",478              "clicks": 2479            }480          ],481          "read": true,482          "user_title": null,483          "bookmarked": false,484          "actions_summary": [],485          "moderator": false,486          "admin": false,487          "staff": false,488          "user_id": 27648,489          "hidden": false,490          "trust_level": 2,491          "deleted_at": null,492          "user_deleted": false,493          "edit_reason": null,494          "can_view_edit_history": true,495          "wiki": false,496          "post_url": "/t/transformer-lm-in-pytorch-documentation-performs-poorly/97691/1",497          "can_accept_answer": false,498          "can_unaccept_answer": false,499          "accepted_answer": false,500          "topic_accepted_answer": null,501          "can_vote": false502        }503      ],504      "stream": [505        232367506      ]507    },508    "timeline_lookup": [509      [510        1,511        1853512      ]513    ],514    "suggested_topics": [515      {516        "fancy_title": "Initial D_KL loss is high and going down really slow",517        "id": 217877,518        "title": "Initial D_KL loss is high and going down really slow",519        "slug": "initial-d-kl-loss-is-high-and-going-down-really-slow",520        "posts_count": 1,521        "reply_count": 0,522        "highest_post_number": 1,523        "image_url": null,524        "created_at": "2025-03-15T12:04:06.828Z",525        "last_posted_at": "2025-03-15T12:04:06.870Z",526        "bumped": true,527        "bumped_at": "2025-03-15T12:07:55.407Z",528        "archetype": "regular",529        "unseen": false,530        "pinned": false,531        "unpinned": null,532        "visible": true,533        "closed": false,534        "archived": false,535        "bookmarked": null,536        "liked": null,537        "tags_descriptions": {},538        "like_count": 0,539        "views": 32,540        "category_id": 8,541        "featured_link": null,542        "has_accepted_answer": false,543        "posters": [544          {545            "extras": "latest single",546            "description": "Original Poster, Most Recent Poster",547            "user": {548              "id": 83290,549              "username": "User_Name",550              "name": "User Name",551              "avatar_template": "/user_avatar/discuss.pytorch.org/user_name/{size}/76177_2.png",552              "trust_level": 0553            }554          }555        ]556      },557      {558        "fancy_title": "What&rsquo;s a good replacement for torchtext?",559        "id": 214345,560        "title": "What's a good replacement for torchtext?",561        "slug": "whats-a-good-replacement-for-torchtext",562        "posts_count": 3,563        "reply_count": 1,564        "highest_post_number": 3,565        "image_url": null,566        "created_at": "2024-12-18T00:55:32.594Z",567        "last_posted_at": "2025-04-02T01:11:46.934Z",568        "bumped": true,569        "bumped_at": "2025-04-02T01:11:46.934Z",570        "archetype": "regular",571        "unseen": false,572        "pinned": false,573        "unpinned": null,574        "visible": true,575        "closed": false,576        "archived": false,577        "bookmarked": null,578        "liked": null,579        "tags_descriptions": {},580        "like_count": 2,581        "views": 983,582        "category_id": 8,583        "featured_link": null,584        "has_accepted_answer": false,585        "posters": [586          {587            "extras": "latest",588            "description": "Original Poster, Most Recent Poster",589            "user": {590              "id": 1438,591              "username": "vdw",592              "name": "Chris",593              "avatar_template": "/user_avatar/discuss.pytorch.org/vdw/{size}/10074_2.png",594              "trust_level": 2595            }596          },597          {598            "extras": null,599            "description": "Frequent Poster",600            "user": {601              "id": 83569,602              "username": "Aaron_M",603              "name": "Aaron Masino",604              "avatar_template": "/user_avatar/discuss.pytorch.org/aaron_m/{size}/76431_2.png",605              "trust_level": 0606            }607          }608        ]609      },610      {611        "fancy_title": "Understanding logits in GPT2",612        "id": 213865,613        "title": "Understanding logits in GPT2",614        "slug": "understanding-logits-in-gpt2",615        "posts_count": 1,616        "reply_count": 0,617        "highest_post_number": 1,618        "image_url": null,619        "created_at": "2024-12-05T15:14:26.136Z",620        "last_posted_at": "2024-12-05T15:14:26.194Z",621        "bumped": true,622        "bumped_at": "2024-12-05T15:14:26.194Z",623        "archetype": "regular",624        "unseen": false,625        "pinned": false,626        "unpinned": null,627        "visible": true,628        "closed": false,629        "archived": false,630        "bookmarked": null,631        "liked": null,632        "tags_descriptions": {},633        "like_count": 0,634        "views": 200,635        "category_id": 8,636        "featured_link": null,637        "has_accepted_answer": false,638        "posters": [639          {640            "extras": "latest single",641            "description": "Original Poster, Most Recent Poster",642            "user": {643              "id": 81335,644              "username": "firolommones3",645              "name": "",646              "avatar_template": "/letter_avatar_proxy/v4/letter/f/b2d939/{size}.png",647              "trust_level": 0648            }649          }650        ]651      },652      {653        "fancy_title": "Need help with Recurrent lstms",654        "id": 215195,655        "title": "Need help with Recurrent lstms",656        "slug": "need-help-with-recurrent-lstms",657        "posts_count": 1,658        "reply_count": 0,659        "highest_post_number": 1,660        "image_url": "https://discuss.pytorch.org/uploads/default/original/3X/b/4/b499c6589da3192d332d676dc1096093ba3df607.jpeg",661        "created_at": "2025-01-10T08:06:38.992Z",662        "last_posted_at": "2025-01-10T08:06:39.052Z",663        "bumped": true,664        "bumped_at": "2025-01-10T08:40:23.002Z",665        "archetype": "regular",666        "unseen": false,667        "pinned": false,668        "unpinned": null,669        "visible": true,670        "closed": false,671        "archived": false,672        "bookmarked": null,673        "liked": null,674        "tags_descriptions": {},675        "like_count": 0,676        "views": 28,677        "category_id": 8,678        "featured_link": null,679        "has_accepted_answer": false,680        "posters": [681          {682            "extras": "latest single",683            "description": "Original Poster, Most Recent Poster",684            "user": {685              "id": 81982,686              "username": "MD_Shahadat_Hossain",687              "name": "MD. Shahadat Hossain Shahal",688              "avatar_template": "/user_avatar/discuss.pytorch.org/md_shahadat_hossain/{size}/75009_2.png",689              "trust_level": 0690            }691          }692        ]693      },694      {695        "fancy_title": "Using a bidirectional nn.GRU Gated Recurrent Unit understand forwarding process",696        "id": 220764,697        "title": "Using a bidirectional nn.GRU Gated Recurrent Unit understand forwarding process",698        "slug": "using-a-bidirectional-nn-gru-gated-recurrent-unit-understand-forwarding-process",699        "posts_count": 1,700        "reply_count": 0,701        "highest_post_number": 1,702        "image_url": null,703        "created_at": "2025-06-12T13:34:08.994Z",704        "last_posted_at": "2025-06-12T13:34:09.036Z",705        "bumped": true,706        "bumped_at": "2025-06-12T13:34:09.036Z",707        "archetype": "regular",708        "unseen": false,709        "pinned": false,710        "unpinned": null,711        "visible": true,712        "closed": false,713        "archived": false,714        "bookmarked": null,715        "liked": null,716        "tags_descriptions": {},717        "like_count": 0,718        "views": 18,719        "category_id": 8,720        "featured_link": null,721        "has_accepted_answer": false,722        "posters": [723          {724            "extras": "latest single",725            "description": "Original Poster, Most Recent Poster",726            "user": {727              "id": 84675,728              "username": "Chefkoch",729              "name": "Chefkoch",730              "avatar_template": "/letter_avatar_proxy/v4/letter/c/977dab/{size}.png",731              "trust_level": 0732            }733          }734        ]735      }736    ],737    "tags_descriptions": {},738    "fancy_title": "Transformer LM in Pytorch Documentation performs poorly",739    "id": 97691,740    "title": "Transformer LM in Pytorch Documentation performs poorly",741    "posts_count": 1,742    "created_at": "2020-09-29T06:10:59.838Z",743    "views": 793,744    "reply_count": 0,745    "like_count": 0,746    "last_posted_at": "2020-09-29T06:10:59.893Z",747    "visible": true,748    "closed": false,749    "archived": false,750    "has_summary": false,751    "archetype": "regular",752    "slug": "transformer-lm-in-pytorch-documentation-performs-poorly",753    "category_id": 8,754    "word_count": 293,755    "deleted_at": null,756    "user_id": 27648,757    "featured_link": null,758    "pinned_globally": false,759    "pinned_at": null,760    "pinned_until": null,761    "image_url": null,762    "slow_mode_seconds": 0,763    "draft": null,764    "draft_key": "topic_97691",765    "draft_sequence": null,766    "unpinned": null,767    "pinned": false,768    "current_post_number": 1,769    "highest_post_number": 1,770    "deleted_by": null,771    "actions_summary": [772      {773        "id": 4,774        "count": 0,775        "hidden": false,776        "can_act": false777      },778      {779        "id": 8,780        "count": 0,781        "hidden": false,782        "can_act": false783      },784      {785        "id": 10,786        "count": 0,787        "hidden": false,788        "can_act": false789      },790      {791        "id": 7,792        "count": 0,793        "hidden": false,794        "can_act": false795      }796    ],797    "chunk_size": 20,798    "bookmarked": false,799    "topic_timer": null,800    "message_bus_last_id": 0,801    "participant_count": 1,802    "show_read_indicator": false,803    "thumbnails": null,804    "slow_mode_enabled_until": null,805    "can_vote": false,806    "vote_count": 0,807    "user_voted": false,808    "discourse_zendesk_plugin_zendesk_id": null,809    "discourse_zendesk_plugin_zendesk_url": "https://your-url.zendesk.com/agent/tickets/",810    "details": {811      "can_edit": false,812      "notification_level": 1,813      "participants": [814        {815          "id": 27648,816          "username": "Hari_Krishnan",817          "name": "Hari Krishnan",818          "avatar_template": "/user_avatar/discuss.pytorch.org/hari_krishnan/{size}/16377_2.png",819          "post_count": 1,820          "primary_group_name": null,821          "flair_name": null,822          "flair_url": null,823          "flair_color": null,824          "flair_bg_color": null,825          "flair_group_id": null,826          "trust_level": 2827        }828      ],829      "created_by": {830        "id": 27648,831        "username": "Hari_Krishnan",832        "name": "Hari Krishnan",833        "avatar_template": "/user_avatar/discuss.pytorch.org/hari_krishnan/{size}/16377_2.png"834      },835      "last_poster": {836        "id": 27648,837        "username": "Hari_Krishnan",838        "name": "Hari Krishnan",839        "avatar_template": "/user_avatar/discuss.pytorch.org/hari_krishnan/{size}/16377_2.png"840      },841      "links": [842        {843          "url": "https://discuss.pytorch.org/t/transformer-model-of-language-model-example-performs-worse-much/87448",844          "title": "Transformer model of Language Model Example performs worse much",845          "internal": true,846          "attachment": false,847          "reflection": false,848          "clicks": 8,849          "user_id": 27648,850          "domain": "discuss.pytorch.org",851          "root_domain": "pytorch.org"852        },853        {854          "url": "https://pytorch.org/tutorials/beginner/transformer_tutorial.html",855          "title": "Sequence-to-Sequence Modeling with nn.Transformer and TorchText — PyTorch Tutorials 1.6.0 documentation",856          "internal": false,857          "attachment": false,858          "reflection": false,859          "clicks": 2,860          "user_id": 27648,861          "domain": "pytorch.org",862          "root_domain": "pytorch.org"863        }864      ]865    },866    "bookmarks": []867  },868  {869    "post_stream": {870      "posts": [871        {872          "id": 13710,873          "name": "ronrest",874          "username": "ronrest",875          "avatar_template": "/user_avatar/discuss.pytorch.org/ronrest/{size}/1134_2.png",876          "created_at": "2017-07-30T09:54:51.066Z",877          "cooked": "<p>In the <a href=\"http://pytorch.org/docs/master/nn.html#torch.nn.LSTM\" rel=\"noopener nofollow ugc\">documentation for LSTM</a>, for the <code>dropout</code> argument, it states:</p>\n<blockquote>\n<p>introduces a dropout layer on the outputs of each RNN layer except the last layer</p>\n</blockquote>\n<p>I just want to clarify what is meant by “everything except the last layer”.</p>\n<p>Below I have an image of two possible options for the meaning.</p>\n<ul>\n<li><strong>Option 1:</strong>  The final cell is the one that does not have dropout applied for the output.</li>\n<li><strong>Option 2:</strong>  In a multi-layer LSTM, all the connections between layers have dropout applied, except the very top layer.  So a single layer LSTM would not have any dropout applied.</li>\n</ul>\n<p><div class=\"lightbox-wrapper\"><a class=\"lightbox\" href=\"https://discuss.pytorch.org/uploads/default/original/2X/6/62f94ceee433b693ef73be231f51ae4291e53880.png\" data-download-href=\"https://discuss.pytorch.org/uploads/default/62f94ceee433b693ef73be231f51ae4291e53880\" title=\"diagram.png\"><img src=\"https://discuss.pytorch.org/uploads/default/optimized/2X/6/62f94ceee433b693ef73be231f51ae4291e53880_2_288x500.png\" width=\"288\" height=\"500\" srcset=\"https://discuss.pytorch.org/uploads/default/optimized/2X/6/62f94ceee433b693ef73be231f51ae4291e53880_2_288x500.png, https://discuss.pytorch.org/uploads/default/original/2X/6/62f94ceee433b693ef73be231f51ae4291e53880.png 1.5x, https://discuss.pytorch.org/uploads/default/original/2X/6/62f94ceee433b693ef73be231f51ae4291e53880.png 2x\" data-dominant-color=\"C8C7AC\"><div class=\"meta\"><svg class=\"fa d-icon d-icon-far-image svg-icon\" aria-hidden=\"true\"><use href=\"#far-image\"></use></svg><span class=\"filename\">diagram.png</span><span class=\"informations\">301×521 27.4 KB</span><svg class=\"fa d-icon d-icon-discourse-expand svg-icon\" aria-hidden=\"true\"><use href=\"#discourse-expand\"></use></svg></div></a></div></p>\n<p>I tried looking at the source code for LSTM and RNNBase, but I can’t figure out how it is being applied.</p>",878          "post_number": 1,879          "post_type": 1,880          "posts_count": 5,881          "updated_at": "2017-07-30T09:54:51.066Z",882          "reply_count": 0,883          "reply_to_post_number": null,884          "quote_count": 0,885          "incoming_link_count": 4121,886          "reads": 299,887          "readers_count": 298,888          "score": 20701.8,889          "yours": false,890          "topic_id": 5588,891          "topic_slug": "lstm-dropout-clarification-of-last-layer",892          "display_username": "ronrest",893          "primary_group_name": null,894          "flair_name": null,895          "flair_url": null,896          "flair_bg_color": null,897          "flair_color": null,898          "flair_group_id": null,899          "badges_granted": [],900          "version": 1,901          "can_edit": false,902          "can_delete": false,903          "can_recover": false,904          "can_see_hidden_post": false,905          "can_wiki": false,906          "link_counts": [907            {908              "url": "http://pytorch.org/docs/master/nn.html#torch.nn.LSTM",909              "internal": false,910              "reflection": false,911              "title": "torch.nn — PyTorch master documentation",912              "clicks": 140913            },914            {915              "url": "https://discuss.pytorch.org/uploads/default/original/2X/6/62f94ceee433b693ef73be231f51ae4291e53880.png",916              "internal": true,917              "reflection": false,918              "clicks": 0919            },920            {921              "url": "https://discuss.pytorch.org/t/dropout-in-lstm/7784/5",922              "internal": true,923              "reflection": true,924              "title": "Dropout in LSTM",925              "clicks": 11926            },927            {928              "url": "https://discuss.pytorch.org/t/seeking-for-clarification-of-dropout-in-rnn-lstm/13963",929              "internal": true,930              "reflection": true,931              "title": "Seeking for clarification of dropout in RNN/LSTM",932              "clicks": 3933            }934          ],935          "read": true,936          "user_title": null,937          "bookmarked": false,938          "actions_summary": [939            {940              "id": 2,941              "count": 2942            }943          ],944          "moderator": false,945          "admin": false,946          "staff": false,947          "user_id": 2559,948          "hidden": false,949          "trust_level": 0,950          "deleted_at": null,951          "user_deleted": false,952          "edit_reason": null,953          "can_view_edit_history": true,954          "wiki": false,955          "post_url": "/t/lstm-dropout-clarification-of-last-layer/5588/1",956          "can_accept_answer": false,957          "can_unaccept_answer": false,958          "accepted_answer": false,959          "topic_accepted_answer": null,960          "can_vote": false961        },962        {963          "id": 13711,964          "name": "",965          "username": "dgriff",966          "avatar_template": "/user_avatar/discuss.pytorch.org/dgriff/{size}/1361_2.png",967          "created_at": "2017-07-30T10:52:56.300Z",968          "cooked": "<p>LSTM is a rolled up LSTMCell and to think of each layer as Performing one LSTMCell action and that should help make sense of it. Number layers being the number of time steps performed</p>",969          "post_number": 2,970          "post_type": 1,971          "posts_count": 5,972          "updated_at": "2017-07-30T10:52:56.300Z",973          "reply_count": 0,974          "reply_to_post_number": null,975          "quote_count": 0,976          "incoming_link_count": 14,977          "reads": 275,978          "readers_count": 274,979          "score": 125.0,980          "yours": false,981          "topic_id": 5588,982          "topic_slug": "lstm-dropout-clarification-of-last-layer",983          "display_username": "",984          "primary_group_name": null,985          "flair_name": null,986          "flair_url": null,987          "flair_bg_color": null,988          "flair_color": null,989          "flair_group_id": null,990          "badges_granted": [],991          "version": 1,992          "can_edit": false,993          "can_delete": false,994          "can_recover": false,995          "can_see_hidden_post": false,996          "can_wiki": false,997          "read": true,998          "user_title": null,999          "bookmarked": false,1000          "actions_summary": [],1001          "moderator": false,1002          "admin": false,1003          "staff": false,1004          "user_id": 1635,1005          "hidden": false,1006          "trust_level": 2,1007          "deleted_at": null,1008          "user_deleted": false,1009          "edit_reason": null,1010          "can_view_edit_history": true,1011          "wiki": false,1012          "post_url": "/t/lstm-dropout-clarification-of-last-layer/5588/2",1013          "can_accept_answer": false,1014          "can_unaccept_answer": false,1015          "accepted_answer": false,1016          "topic_accepted_answer": null1017        },1018        {1019          "id": 13714,1020          "name": "Thomas V",1021          "username": "tom",1022          "avatar_template": "/user_avatar/discuss.pytorch.org/tom/{size}/3162_2.png",1023          "created_at": "2017-07-30T13:02:08.344Z",1024          "cooked": "<p>Hello,</p>\n<p>from both the text you quoted and the code, I’d say likely option 2. This is my understanding of layers.</p>\n<p>If you believe that, after disabling cudnn, this line is relevant, it might also give a hint.<br>\n<aside class=\"onebox githubblob\">\n  <header class=\"source\">\n      <a href=\"https://github.com/pytorch/pytorch/blob/master/torch/nn/_functions/rnn.py#L91\" target=\"_blank\" rel=\"nofollow noopener\">github.com</a>\n  </header>\n  <article class=\"onebox-body\">\n    <h4><a href=\"https://github.com/pytorch/pytorch/blob/master/torch/nn/_functions/rnn.py#L91\" target=\"_blank\" rel=\"nofollow noopener\">pytorch/pytorch/blob/master/torch/nn/_functions/rnn.py#L91</a></h4>\n<pre class=\"onebox\"><code class=\"lang-py\"><ol class=\"start lines\" start=\"81\" style=\"counter-reset: li-counter 80 ;\">\n<li>    for j, inner in enumerate(inners):</li>\n<li>        l = i * num_directions + j</li>\n<li>\n</li>\n<li>        hy, output = inner(input, hidden[l], weight[l])</li>\n<li>        next_hidden.append(hy)</li>\n<li>        all_output.append(output)</li>\n<li>\n</li>\n<li>    input = torch.cat(all_output, input.dim() - 1)</li>\n<li>\n</li>\n<li>    if dropout != 0 and i &lt; num_layers - 1:</li>\n<li class=\"selected\">        input = F.dropout(input, p=dropout, training=train, inplace=False)</li>\n<li>\n</li>\n<li>if lstm:</li>\n<li>    next_h, next_c = zip(*next_hidden)</li>\n<li>    next_hidden = (</li>\n<li>        torch.cat(next_h, 0).view(total_layers, *next_h[0].size()),</li>\n<li>        torch.cat(next_c, 0).view(total_layers, *next_c[0].size())</li>\n<li>    )</li>\n<li>else:</li>\n<li>    next_hidden = torch.cat(next_hidden, 0).view(</li>\n<li>        total_layers, *next_hidden[0].size())</li>\n</ol></code></pre>\n\n\n  </article>\n  <div class=\"onebox-metadata\">\n    \n    \n  </div>\n  <div style=\"clear: both\"></div>\n</aside>\n<br>\nMy understanding of this code is that it does a single timestep (your horizontal axis) for many layers.</p>\n<p>Best regards</p>\n<p>Thomas</p>",1025          "post_number": 3,1026          "post_type": 1,1027          "posts_count": 5,1028          "updated_at": "2017-07-30T13:02:08.344Z",1029          "reply_count": 1,1030          "reply_to_post_number": null,1031          "quote_count": 0,1032          "incoming_link_count": 69,1033          "reads": 274,1034          "readers_count": 273,1035          "score": 451.8,1036          "yours": false,1037          "topic_id": 5588,1038          "topic_slug": "lstm-dropout-clarification-of-last-layer",1039          "display_username": "Thomas V",1040          "primary_group_name": null,1041          "flair_name": null,1042          "flair_url": null,1043          "flair_bg_color": null,1044          "flair_color": null,1045          "flair_group_id": null,1046          "badges_granted": [],1047          "version": 1,1048          "can_edit": false,1049          "can_delete": false,1050          "can_recover": false,1051          "can_see_hidden_post": false,1052          "can_wiki": false,1053          "link_counts": [1054            {1055              "url": "https://github.com/pytorch/pytorch/blob/master/torch/nn/_functions/rnn.py#L91",1056              "internal": false,1057              "reflection": false,1058              "title": "pytorch/rnn.py at master · pytorch/pytorch · GitHub",1059              "clicks": 931060            }1061          ],1062          "read": true,1063          "user_title": null,1064          "bookmarked": false,1065          "actions_summary": [1066            {1067              "id": 2,1068              "count": 21069            }1070          ],1071          "moderator": false,1072          "admin": false,1073          "staff": false,1074          "user_id": 616,1075          "hidden": false,1076          "trust_level": 2,1077          "deleted_at": null,1078          "user_deleted": false,1079          "edit_reason": null,1080          "can_view_edit_history": true,1081          "wiki": false,1082          "post_url": "/t/lstm-dropout-clarification-of-last-layer/5588/3",1083          "can_accept_answer": false,1084          "can_unaccept_answer": false,1085          "accepted_answer": false,1086          "topic_accepted_answer": null1087        },1088        {1089          "id": 13718,1090          "name": "",1091          "username": "dgriff",1092          "avatar_template": "/user_avatar/discuss.pytorch.org/dgriff/{size}/1361_2.png",1093          "created_at": "2017-07-30T14:01:28.897Z",1094          "cooked": "<p>Time steps I’m referring from going from each element in the input. Like if you fed a sentence as sequence the order of the words in that sentence matters. Hence why the pytorch docs say:</p>\n<p>“For each element in the input sequence, each layer computes the following function”</p>\n<p>Nn.LSTM is a window LSTM Aka rolled up LSTM</p>\n<p>And would be option 2</p>\n<p>Which sorry I misread you Tom thought you were disagreeing with that but read incorrectly and what I wrote was wrong it is multilayer for each time step😲</p>",1095          "post_number": 4,1096          "post_type": 1,1097          "posts_count": 5,1098          "updated_at": "2017-07-30T14:33:00.467Z",1099          "reply_count": 0,1100          "reply_to_post_number": null,1101          "quote_count": 0,1102          "incoming_link_count": 24,1103          "reads": 234,1104          "readers_count": 233,1105          "score": 166.8,1106          "yours": false,1107          "topic_id": 5588,1108          "topic_slug": "lstm-dropout-clarification-of-last-layer",1109          "display_username": "",1110          "primary_group_name": null,1111          "flair_name": null,1112          "flair_url": null,1113          "flair_bg_color": null,1114          "flair_color": null,1115          "flair_group_id": null,1116          "badges_granted": [],1117          "version": 3,1118          "can_edit": false,1119          "can_delete": false,1120          "can_recover": false,1121          "can_see_hidden_post": false,1122          "can_wiki": false,1123          "read": true,1124          "user_title": null,1125          "bookmarked": false,1126          "actions_summary": [],1127          "moderator": false,1128          "admin": false,1129          "staff": false,1130          "user_id": 1635,1131          "hidden": false,1132          "trust_level": 2,1133          "deleted_at": null,1134          "user_deleted": false,1135          "edit_reason": null,1136          "can_view_edit_history": true,1137          "wiki": false,1138          "post_url": "/t/lstm-dropout-clarification-of-last-layer/5588/4",1139          "can_accept_answer": false,1140          "can_unaccept_answer": false,1141          "accepted_answer": false,1142          "topic_accepted_answer": null1143        },1144        {1145          "id": 232369,1146          "name": "CODE",1147          "username": "STU",1148          "avatar_template": "/user_avatar/discuss.pytorch.org/stu/{size}/16640_2.png",1149          "created_at": "2020-09-29T06:37:57.158Z",1150          "cooked": "<p>Hi, if it means the Option-1, why does the num_layers must great than 1?</p>",1151          "post_number": 5,1152          "post_type": 1,1153          "posts_count": 5,1154          "updated_at": "2020-09-29T06:37:57.158Z",1155          "reply_count": 0,1156          "reply_to_post_number": 3,1157          "quote_count": 0,1158          "incoming_link_count": 8,1159          "reads": 32,1160          "readers_count": 31,1161          "score": 46.4,1162          "yours": false,1163          "topic_id": 5588,1164          "topic_slug": "lstm-dropout-clarification-of-last-layer",1165          "display_username": "CODE",1166          "primary_group_name": null,1167          "flair_name": null,1168          "flair_url": null,1169          "flair_bg_color": null,1170          "flair_color": null,1171          "flair_group_id": null,1172          "badges_granted": [],1173          "version": 1,1174          "can_edit": false,1175          "can_delete": false,1176          "can_recover": false,1177          "can_see_hidden_post": false,1178          "can_wiki": false,1179          "read": true,1180          "user_title": null,1181          "reply_to_user": {1182            "id": 616,1183            "username": "tom",1184            "name": "Thomas V",1185            "avatar_template": "/user_avatar/discuss.pytorch.org/tom/{size}/3162_2.png"1186          },1187          "bookmarked": false,1188          "actions_summary": [],1189          "moderator": false,1190          "admin": false,1191          "staff": false,1192          "user_id": 15192,1193          "hidden": false,1194          "trust_level": 1,1195          "deleted_at": null,1196          "user_deleted": false,1197          "edit_reason": null,1198          "can_view_edit_history": true,1199          "wiki": false,1200          "post_url": "/t/lstm-dropout-clarification-of-last-layer/5588/5",

Showing the first 1,200 of 65575 lines. Download the file for the rest.