CoolFace
Apppublic

romiio/Text-summarization

sourceHugging Faceotherupdated 3y agoView on Hugging Face
0likes
Text_Summarization.ipynb1258 linesDownload Raw Back to root
1{2  "nbformat": 4,3  "nbformat_minor": 0,4  "metadata": {5    "colab": {6      "provenance": []7    },8    "kernelspec": {9      "name": "python3",10      "display_name": "Python 3"11    },12    "language_info": {13      "name": "python"14    },15    "widgets": {16      "application/vnd.jupyter.widget-state+json": {17        "1427097f4c03437ba6545632c57b0027": {18          "model_module": "@jupyter-widgets/controls",19          "model_name": "HBoxModel",20          "model_module_version": "1.5.0",21          "state": {22            "_dom_classes": [],23            "_model_module": "@jupyter-widgets/controls",24            "_model_module_version": "1.5.0",25            "_model_name": "HBoxModel",26            "_view_count": null,27            "_view_module": "@jupyter-widgets/controls",28            "_view_module_version": "1.5.0",29            "_view_name": "HBoxView",30            "box_style": "",31            "children": [32              "IPY_MODEL_46fcd96cdb53432a91a36775693d3d1f",33              "IPY_MODEL_124ec5caf31b48b1a890bd6dbeabd51b",34              "IPY_MODEL_a748ec3f25534cedab90d9953d43d85b"35            ],36            "layout": "IPY_MODEL_cd070b0e2c8b48b3b02848cb89b9cded"37          }38        },39        "46fcd96cdb53432a91a36775693d3d1f": {40          "model_module": "@jupyter-widgets/controls",41          "model_name": "HTMLModel",42          "model_module_version": "1.5.0",43          "state": {44            "_dom_classes": [],45            "_model_module": "@jupyter-widgets/controls",46            "_model_module_version": "1.5.0",47            "_model_name": "HTMLModel",48            "_view_count": null,49            "_view_module": "@jupyter-widgets/controls",50            "_view_module_version": "1.5.0",51            "_view_name": "HTMLView",52            "description": "",53            "description_tooltip": null,54            "layout": "IPY_MODEL_f2a57bcaa5d545dc830e6b3d7a6c921e",55            "placeholder": "​",56            "style": "IPY_MODEL_d442f282871a490b905f2c4562c107db",57            "value": "Downloading builder script: "58          }59        },60        "124ec5caf31b48b1a890bd6dbeabd51b": {61          "model_module": "@jupyter-widgets/controls",62          "model_name": "FloatProgressModel",63          "model_module_version": "1.5.0",64          "state": {65            "_dom_classes": [],66            "_model_module": "@jupyter-widgets/controls",67            "_model_module_version": "1.5.0",68            "_model_name": "FloatProgressModel",69            "_view_count": null,70            "_view_module": "@jupyter-widgets/controls",71            "_view_module_version": "1.5.0",72            "_view_name": "ProgressView",73            "bar_style": "success",74            "description": "",75            "description_tooltip": null,76            "layout": "IPY_MODEL_ae5a78baca7647afb9f095c99e6285ca",77            "max": 2169,78            "min": 0,79            "orientation": "horizontal",80            "style": "IPY_MODEL_4a726f29235948fba545f9bd684c9312",81            "value": 216982          }83        },84        "a748ec3f25534cedab90d9953d43d85b": {85          "model_module": "@jupyter-widgets/controls",86          "model_name": "HTMLModel",87          "model_module_version": "1.5.0",88          "state": {89            "_dom_classes": [],90            "_model_module": "@jupyter-widgets/controls",91            "_model_module_version": "1.5.0",92            "_model_name": "HTMLModel",93            "_view_count": null,94            "_view_module": "@jupyter-widgets/controls",95            "_view_module_version": "1.5.0",96            "_view_name": "HTMLView",97            "description": "",98            "description_tooltip": null,99            "layout": "IPY_MODEL_630c7d9853da46aca7cf874d61545e3e",100            "placeholder": "​",101            "style": "IPY_MODEL_46e64861f65446eab869bb844f53e470",102            "value": " 5.65k/? [00:00&lt;00:00, 270kB/s]"103          }104        },105        "cd070b0e2c8b48b3b02848cb89b9cded": {106          "model_module": "@jupyter-widgets/base",107          "model_name": "LayoutModel",108          "model_module_version": "1.2.0",109          "state": {110            "_model_module": "@jupyter-widgets/base",111            "_model_module_version": "1.2.0",112            "_model_name": "LayoutModel",113            "_view_count": null,114            "_view_module": "@jupyter-widgets/base",115            "_view_module_version": "1.2.0",116            "_view_name": "LayoutView",117            "align_content": null,118            "align_items": null,119            "align_self": null,120            "border": null,121            "bottom": null,122            "display": null,123            "flex": null,124            "flex_flow": null,125            "grid_area": null,126            "grid_auto_columns": null,127            "grid_auto_flow": null,128            "grid_auto_rows": null,129            "grid_column": null,130            "grid_gap": null,131            "grid_row": null,132            "grid_template_areas": null,133            "grid_template_columns": null,134            "grid_template_rows": null,135            "height": null,136            "justify_content": null,137            "justify_items": null,138            "left": null,139            "margin": null,140            "max_height": null,141            "max_width": null,142            "min_height": null,143            "min_width": null,144            "object_fit": null,145            "object_position": null,146            "order": null,147            "overflow": null,148            "overflow_x": null,149            "overflow_y": null,150            "padding": null,151            "right": null,152            "top": null,153            "visibility": null,154            "width": null155          }156        },157        "f2a57bcaa5d545dc830e6b3d7a6c921e": {158          "model_module": "@jupyter-widgets/base",159          "model_name": "LayoutModel",160          "model_module_version": "1.2.0",161          "state": {162            "_model_module": "@jupyter-widgets/base",163            "_model_module_version": "1.2.0",164            "_model_name": "LayoutModel",165            "_view_count": null,166            "_view_module": "@jupyter-widgets/base",167            "_view_module_version": "1.2.0",168            "_view_name": "LayoutView",169            "align_content": null,170            "align_items": null,171            "align_self": null,172            "border": null,173            "bottom": null,174            "display": null,175            "flex": null,176            "flex_flow": null,177            "grid_area": null,178            "grid_auto_columns": null,179            "grid_auto_flow": null,180            "grid_auto_rows": null,181            "grid_column": null,182            "grid_gap": null,183            "grid_row": null,184            "grid_template_areas": null,185            "grid_template_columns": null,186            "grid_template_rows": null,187            "height": null,188            "justify_content": null,189            "justify_items": null,190            "left": null,191            "margin": null,192            "max_height": null,193            "max_width": null,194            "min_height": null,195            "min_width": null,196            "object_fit": null,197            "object_position": null,198            "order": null,199            "overflow": null,200            "overflow_x": null,201            "overflow_y": null,202            "padding": null,203            "right": null,204            "top": null,205            "visibility": null,206            "width": null207          }208        },209        "d442f282871a490b905f2c4562c107db": {210          "model_module": "@jupyter-widgets/controls",211          "model_name": "DescriptionStyleModel",212          "model_module_version": "1.5.0",213          "state": {214            "_model_module": "@jupyter-widgets/controls",215            "_model_module_version": "1.5.0",216            "_model_name": "DescriptionStyleModel",217            "_view_count": null,218            "_view_module": "@jupyter-widgets/base",219            "_view_module_version": "1.2.0",220            "_view_name": "StyleView",221            "description_width": ""222          }223        },224        "ae5a78baca7647afb9f095c99e6285ca": {225          "model_module": "@jupyter-widgets/base",226          "model_name": "LayoutModel",227          "model_module_version": "1.2.0",228          "state": {229            "_model_module": "@jupyter-widgets/base",230            "_model_module_version": "1.2.0",231            "_model_name": "LayoutModel",232            "_view_count": null,233            "_view_module": "@jupyter-widgets/base",234            "_view_module_version": "1.2.0",235            "_view_name": "LayoutView",236            "align_content": null,237            "align_items": null,238            "align_self": null,239            "border": null,240            "bottom": null,241            "display": null,242            "flex": null,243            "flex_flow": null,244            "grid_area": null,245            "grid_auto_columns": null,246            "grid_auto_flow": null,247            "grid_auto_rows": null,248            "grid_column": null,249            "grid_gap": null,250            "grid_row": null,251            "grid_template_areas": null,252            "grid_template_columns": null,253            "grid_template_rows": null,254            "height": null,255            "justify_content": null,256            "justify_items": null,257            "left": null,258            "margin": null,259            "max_height": null,260            "max_width": null,261            "min_height": null,262            "min_width": null,263            "object_fit": null,264            "object_position": null,265            "order": null,266            "overflow": null,267            "overflow_x": null,268            "overflow_y": null,269            "padding": null,270            "right": null,271            "top": null,272            "visibility": null,273            "width": null274          }275        },276        "4a726f29235948fba545f9bd684c9312": {277          "model_module": "@jupyter-widgets/controls",278          "model_name": "ProgressStyleModel",279          "model_module_version": "1.5.0",280          "state": {281            "_model_module": "@jupyter-widgets/controls",282            "_model_module_version": "1.5.0",283            "_model_name": "ProgressStyleModel",284            "_view_count": null,285            "_view_module": "@jupyter-widgets/base",286            "_view_module_version": "1.2.0",287            "_view_name": "StyleView",288            "bar_color": null,289            "description_width": ""290          }291        },292        "630c7d9853da46aca7cf874d61545e3e": {293          "model_module": "@jupyter-widgets/base",294          "model_name": "LayoutModel",295          "model_module_version": "1.2.0",296          "state": {297            "_model_module": "@jupyter-widgets/base",298            "_model_module_version": "1.2.0",299            "_model_name": "LayoutModel",300            "_view_count": null,301            "_view_module": "@jupyter-widgets/base",302            "_view_module_version": "1.2.0",303            "_view_name": "LayoutView",304            "align_content": null,305            "align_items": null,306            "align_self": null,307            "border": null,308            "bottom": null,309            "display": null,310            "flex": null,311            "flex_flow": null,312            "grid_area": null,313            "grid_auto_columns": null,314            "grid_auto_flow": null,315            "grid_auto_rows": null,316            "grid_column": null,317            "grid_gap": null,318            "grid_row": null,319            "grid_template_areas": null,320            "grid_template_columns": null,321            "grid_template_rows": null,322            "height": null,323            "justify_content": null,324            "justify_items": null,325            "left": null,326            "margin": null,327            "max_height": null,328            "max_width": null,329            "min_height": null,330            "min_width": null,331            "object_fit": null,332            "object_position": null,333            "order": null,334            "overflow": null,335            "overflow_x": null,336            "overflow_y": null,337            "padding": null,338            "right": null,339            "top": null,340            "visibility": null,341            "width": null342          }343        },344        "46e64861f65446eab869bb844f53e470": {345          "model_module": "@jupyter-widgets/controls",346          "model_name": "DescriptionStyleModel",347          "model_module_version": "1.5.0",348          "state": {349            "_model_module": "@jupyter-widgets/controls",350            "_model_module_version": "1.5.0",351            "_model_name": "DescriptionStyleModel",352            "_view_count": null,353            "_view_module": "@jupyter-widgets/base",354            "_view_module_version": "1.2.0",355            "_view_name": "StyleView",356            "description_width": ""357          }358        }359      }360    }361  },362  "cells": [363    {364      "cell_type": "code",365      "execution_count": 2,366      "metadata": {367        "colab": {368          "base_uri": "https://localhost:8080/"369        },370        "id": "QEED7U1ux-4q",371        "outputId": "9968baa9-a985-4b47-9c35-ff265134d84d"372      },373      "outputs": [374        {375          "output_type": "stream",376          "name": "stdout",377          "text": [378            "/bin/bash: line 1: nvidia-smi: command not found\n"379          ]380        }381      ],382      "source": [383        "!nvidia-smi"384      ]385    },386    {387      "cell_type": "code",388      "source": [389        "!pip install transformers[sentencepiece] datasets sacrebleu rouge_score py7zr -q\n"390      ],391      "metadata": {392        "id": "jSDwTbU4yile"393      },394      "execution_count": 3,395      "outputs": []396    },397    {398      "cell_type": "code",399      "source": [400        "!pip install --upgrade accelerate\n",401        "!pip uninstall -y transformers accelerate\n",402        "!pip install transformers accelerate\n",403        "!pip install transformers"404      ],405      "metadata": {406        "colab": {407          "base_uri": "https://localhost:8080/"408        },409        "id": "3dK8-LNAz2sV",410        "outputId": "a00653df-a6e0-4f11-d277-3af13b15a747"411      },412      "execution_count": 4,413      "outputs": [414        {415          "output_type": "stream",416          "name": "stdout",417          "text": [418            "Requirement already satisfied: accelerate in /usr/local/lib/python3.10/dist-packages (0.26.1)\n",419            "Requirement already satisfied: numpy>=1.17 in /usr/local/lib/python3.10/dist-packages (from accelerate) (1.23.5)\n",420            "Requirement already satisfied: packaging>=20.0 in /usr/local/lib/python3.10/dist-packages (from accelerate) (23.2)\n",421            "Requirement already satisfied: psutil in /usr/local/lib/python3.10/dist-packages (from accelerate) (5.9.5)\n",422            "Requirement already satisfied: pyyaml in /usr/local/lib/python3.10/dist-packages (from accelerate) (6.0.1)\n",423            "Requirement already satisfied: torch>=1.10.0 in /usr/local/lib/python3.10/dist-packages (from accelerate) (2.1.0+cu121)\n",424            "Requirement already satisfied: huggingface-hub in /usr/local/lib/python3.10/dist-packages (from accelerate) (0.20.3)\n",425            "Requirement already satisfied: safetensors>=0.3.1 in /usr/local/lib/python3.10/dist-packages (from accelerate) (0.4.2)\n",426            "Requirement already satisfied: filelock in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (3.13.1)\n",427            "Requirement already satisfied: typing-extensions in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (4.9.0)\n",428            "Requirement already satisfied: sympy in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (1.12)\n",429            "Requirement already satisfied: networkx in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (3.2.1)\n",430            "Requirement already satisfied: jinja2 in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (3.1.3)\n",431            "Requirement already satisfied: fsspec in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (2023.6.0)\n",432            "Requirement already satisfied: triton==2.1.0 in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (2.1.0)\n",433            "Requirement already satisfied: requests in /usr/local/lib/python3.10/dist-packages (from huggingface-hub->accelerate) (2.31.0)\n",434            "Requirement already satisfied: tqdm>=4.42.1 in /usr/local/lib/python3.10/dist-packages (from huggingface-hub->accelerate) (4.66.1)\n",435            "Requirement already satisfied: MarkupSafe>=2.0 in /usr/local/lib/python3.10/dist-packages (from jinja2->torch>=1.10.0->accelerate) (2.1.5)\n",436            "Requirement already satisfied: charset-normalizer<4,>=2 in /usr/local/lib/python3.10/dist-packages (from requests->huggingface-hub->accelerate) (3.3.2)\n",437            "Requirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.10/dist-packages (from requests->huggingface-hub->accelerate) (3.6)\n",438            "Requirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.10/dist-packages (from requests->huggingface-hub->accelerate) (2.0.7)\n",439            "Requirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.10/dist-packages (from requests->huggingface-hub->accelerate) (2024.2.2)\n",440            "Requirement already satisfied: mpmath>=0.19 in /usr/local/lib/python3.10/dist-packages (from sympy->torch>=1.10.0->accelerate) (1.3.0)\n",441            "Found existing installation: transformers 4.37.2\n",442            "Uninstalling transformers-4.37.2:\n",443            "  Successfully uninstalled transformers-4.37.2\n",444            "Found existing installation: accelerate 0.26.1\n",445            "Uninstalling accelerate-0.26.1:\n",446            "  Successfully uninstalled accelerate-0.26.1\n",447            "Collecting transformers\n",448            "  Using cached transformers-4.37.2-py3-none-any.whl (8.4 MB)\n",449            "Collecting accelerate\n",450            "  Using cached accelerate-0.26.1-py3-none-any.whl (270 kB)\n",451            "Requirement already satisfied: filelock in /usr/local/lib/python3.10/dist-packages (from transformers) (3.13.1)\n",452            "Requirement already satisfied: huggingface-hub<1.0,>=0.19.3 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.20.3)\n",453            "Requirement already satisfied: numpy>=1.17 in /usr/local/lib/python3.10/dist-packages (from transformers) (1.23.5)\n",454            "Requirement already satisfied: packaging>=20.0 in /usr/local/lib/python3.10/dist-packages (from transformers) (23.2)\n",455            "Requirement already satisfied: pyyaml>=5.1 in /usr/local/lib/python3.10/dist-packages (from transformers) (6.0.1)\n",456            "Requirement already satisfied: regex!=2019.12.17 in /usr/local/lib/python3.10/dist-packages (from transformers) (2023.12.25)\n",457            "Requirement already satisfied: requests in /usr/local/lib/python3.10/dist-packages (from transformers) (2.31.0)\n",458            "Requirement already satisfied: tokenizers<0.19,>=0.14 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.15.1)\n",459            "Requirement already satisfied: safetensors>=0.4.1 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.4.2)\n",460            "Requirement already satisfied: tqdm>=4.27 in /usr/local/lib/python3.10/dist-packages (from transformers) (4.66.1)\n",461            "Requirement already satisfied: psutil in /usr/local/lib/python3.10/dist-packages (from accelerate) (5.9.5)\n",462            "Requirement already satisfied: torch>=1.10.0 in /usr/local/lib/python3.10/dist-packages (from accelerate) (2.1.0+cu121)\n",463            "Requirement already satisfied: fsspec>=2023.5.0 in /usr/local/lib/python3.10/dist-packages (from huggingface-hub<1.0,>=0.19.3->transformers) (2023.6.0)\n",464            "Requirement already satisfied: typing-extensions>=3.7.4.3 in /usr/local/lib/python3.10/dist-packages (from huggingface-hub<1.0,>=0.19.3->transformers) (4.9.0)\n",465            "Requirement already satisfied: sympy in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (1.12)\n",466            "Requirement already satisfied: networkx in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (3.2.1)\n",467            "Requirement already satisfied: jinja2 in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (3.1.3)\n",468            "Requirement already satisfied: triton==2.1.0 in /usr/local/lib/python3.10/dist-packages (from torch>=1.10.0->accelerate) (2.1.0)\n",469            "Requirement already satisfied: charset-normalizer<4,>=2 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (3.3.2)\n",470            "Requirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (3.6)\n",471            "Requirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (2.0.7)\n",472            "Requirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (2024.2.2)\n",473            "Requirement already satisfied: MarkupSafe>=2.0 in /usr/local/lib/python3.10/dist-packages (from jinja2->torch>=1.10.0->accelerate) (2.1.5)\n",474            "Requirement already satisfied: mpmath>=0.19 in /usr/local/lib/python3.10/dist-packages (from sympy->torch>=1.10.0->accelerate) (1.3.0)\n",475            "Installing collected packages: accelerate, transformers\n",476            "Successfully installed accelerate-0.26.1 transformers-4.37.2\n",477            "Requirement already satisfied: transformers in /usr/local/lib/python3.10/dist-packages (4.37.2)\n",478            "Requirement already satisfied: filelock in /usr/local/lib/python3.10/dist-packages (from transformers) (3.13.1)\n",479            "Requirement already satisfied: huggingface-hub<1.0,>=0.19.3 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.20.3)\n",480            "Requirement already satisfied: numpy>=1.17 in /usr/local/lib/python3.10/dist-packages (from transformers) (1.23.5)\n",481            "Requirement already satisfied: packaging>=20.0 in /usr/local/lib/python3.10/dist-packages (from transformers) (23.2)\n",482            "Requirement already satisfied: pyyaml>=5.1 in /usr/local/lib/python3.10/dist-packages (from transformers) (6.0.1)\n",483            "Requirement already satisfied: regex!=2019.12.17 in /usr/local/lib/python3.10/dist-packages (from transformers) (2023.12.25)\n",484            "Requirement already satisfied: requests in /usr/local/lib/python3.10/dist-packages (from transformers) (2.31.0)\n",485            "Requirement already satisfied: tokenizers<0.19,>=0.14 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.15.1)\n",486            "Requirement already satisfied: safetensors>=0.4.1 in /usr/local/lib/python3.10/dist-packages (from transformers) (0.4.2)\n",487            "Requirement already satisfied: tqdm>=4.27 in /usr/local/lib/python3.10/dist-packages (from transformers) (4.66.1)\n",488            "Requirement already satisfied: fsspec>=2023.5.0 in /usr/local/lib/python3.10/dist-packages (from huggingface-hub<1.0,>=0.19.3->transformers) (2023.6.0)\n",489            "Requirement already satisfied: typing-extensions>=3.7.4.3 in /usr/local/lib/python3.10/dist-packages (from huggingface-hub<1.0,>=0.19.3->transformers) (4.9.0)\n",490            "Requirement already satisfied: charset-normalizer<4,>=2 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (3.3.2)\n",491            "Requirement already satisfied: idna<4,>=2.5 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (3.6)\n",492            "Requirement already satisfied: urllib3<3,>=1.21.1 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (2.0.7)\n",493            "Requirement already satisfied: certifi>=2017.4.17 in /usr/local/lib/python3.10/dist-packages (from requests->transformers) (2024.2.2)\n"494          ]495        }496      ]497    },498    {499      "cell_type": "code",500      "source": [501        "from transformers import pipeline, set_seed\n",502        "from datasets import load_dataset, load_from_disk\n",503        "import matplotlib.pyplot as plt\n",504        "from datasets import load_dataset\n",505        "import pandas as pd\n",506        "from datasets import load_dataset, load_metric\n",507        "\n",508        "from transformers import AutoModelForSeq2SeqLM, AutoTokenizer\n",509        "\n",510        "import nltk\n",511        "from nltk.tokenize import sent_tokenize\n",512        "\n",513        "from tqdm import tqdm\n",514        "import torch\n",515        "\n",516        "nltk.download(\"punkt\")"517      ],518      "metadata": {519        "id": "IKAKTCNR04-u",520        "colab": {521          "base_uri": "https://localhost:8080/"522        },523        "outputId": "a6ecf705-c070-423f-f2de-9f9ecee907bd"524      },525      "execution_count": 1,526      "outputs": [527        {528          "output_type": "stream",529          "name": "stderr",530          "text": [531            "[nltk_data] Downloading package punkt to /root/nltk_data...\n",532            "[nltk_data]   Package punkt is already up-to-date!\n"533          ]534        },535        {536          "output_type": "execute_result",537          "data": {538            "text/plain": [539              "True"540            ]541          },542          "metadata": {},543          "execution_count": 1544        }545      ]546    },547    {548      "cell_type": "code",549      "source": [550        "#Setting up Device, to use GPU\n",551        "from transformers import AutoModelForSeq2SeqLM, AutoTokenizer\n",552        "device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n",553        "device"554      ],555      "metadata": {556        "id": "V_UaMyQs19Ej",557        "colab": {558          "base_uri": "https://localhost:8080/",559          "height": 36560        },561        "outputId": "0d224eeb-15c6-4ff0-d91f-9acb82b932f2"562      },563      "execution_count": 2,564      "outputs": [565        {566          "output_type": "execute_result",567          "data": {568            "text/plain": [569              "'cpu'"570            ],571            "application/vnd.google.colaboratory.intrinsic+json": {572              "type": "string"573            }574          },575          "metadata": {},576          "execution_count": 2577        }578      ]579    },580    {581      "cell_type": "code",582      "source": [583        "# Loade Model and it's Tokenizer\n",584        "model_ckpt = \"google/pegasus-cnn_dailymail\"\n",585        "model_pegasus = AutoModelForSeq2SeqLM.from_pretrained(model_ckpt).to(device)\n",586        "tokenizer = AutoTokenizer.from_pretrained(model_ckpt)\n"587      ],588      "metadata": {589        "id": "q3fmrv0VJcKD",590        "colab": {591          "base_uri": "https://localhost:8080/"592        },593        "outputId": "ebb58bdb-52d0-4a29-f3f2-623db7e8c099"594      },595      "execution_count": 3,596      "outputs": [597        {598          "output_type": "stream",599          "name": "stderr",600          "text": [601            "/usr/local/lib/python3.10/dist-packages/huggingface_hub/utils/_token.py:88: UserWarning: \n",602            "The secret `HF_TOKEN` does not exist in your Colab secrets.\n",603            "To authenticate with the Hugging Face Hub, create a token in your settings tab (https://huggingface.co/settings/tokens), set it as secret in your Google Colab and restart your session.\n",604            "You will be able to reuse this secret in all of your notebooks.\n",605            "Please note that authentication is recommended but still optional to access public models or datasets.\n",606            "  warnings.warn(\n",607            "/usr/local/lib/python3.10/dist-packages/torch/_utils.py:831: UserWarning: TypedStorage is deprecated. It will be removed in the future and UntypedStorage will be the only storage class. This should only matter to you if you are using storages directly.  To access UntypedStorage directly, use tensor.untyped_storage() instead of tensor.storage()\n",608            "  return self.fget.__get__(instance, owner)()\n",609            "Some weights of PegasusForConditionalGeneration were not initialized from the model checkpoint at google/pegasus-cnn_dailymail and are newly initialized: ['model.decoder.embed_positions.weight', 'model.encoder.embed_positions.weight']\n",610            "You should probably TRAIN this model on a down-stream task to be able to use it for predictions and inference.\n"611          ]612        }613      ]614    },615    {616      "cell_type": "code",617      "source": [618        "# Download and Unzip the Data\n",619        "!wget https://github.com/entbappy/Branching-tutorial/raw/master/summarizer-data.zip\n",620        "!unzip summarizer-data.zip"621      ],622      "metadata": {623        "id": "AqQk2EiaaXaS",624        "colab": {625          "base_uri": "https://localhost:8080/"626        },627        "outputId": "ae54331f-043f-4a99-822c-439d3eda8041"628      },629      "execution_count": 8,630      "outputs": [631        {632          "output_type": "stream",633          "name": "stdout",634          "text": [635            "--2024-02-08 23:17:27--  https://github.com/entbappy/Branching-tutorial/raw/master/summarizer-data.zip\n",636            "Resolving github.com (github.com)... 140.82.113.4\n",637            "Connecting to github.com (github.com)|140.82.113.4|:443... connected.\n",638            "HTTP request sent, awaiting response... 302 Found\n",639            "Location: https://raw.githubusercontent.com/entbappy/Branching-tutorial/master/summarizer-data.zip [following]\n",640            "--2024-02-08 23:17:27--  https://raw.githubusercontent.com/entbappy/Branching-tutorial/master/summarizer-data.zip\n",641            "Resolving raw.githubusercontent.com (raw.githubusercontent.com)... 185.199.108.133, 185.199.109.133, 185.199.111.133, ...\n",642            "Connecting to raw.githubusercontent.com (raw.githubusercontent.com)|185.199.108.133|:443... connected.\n",643            "HTTP request sent, awaiting response... 200 OK\n",644            "Length: 7903594 (7.5M) [application/zip]\n",645            "Saving to: ‘summarizer-data.zip.1’\n",646            "\n",647            "summarizer-data.zip 100%[===================>]   7.54M  --.-KB/s    in 0.09s   \n",648            "\n",649            "2024-02-08 23:17:27 (87.4 MB/s) - ‘summarizer-data.zip.1’ saved [7903594/7903594]\n",650            "\n",651            "Archive:  summarizer-data.zip\n",652            "replace samsum-test.csv? [y]es, [n]o, [A]ll, [N]one, [r]ename: "653          ]654        }655      ]656    },657    {658      "cell_type": "code",659      "source": [660        "dataset_samsum = load_from_disk('samsum_dataset')\n",661        "dataset_samsum"662      ],663      "metadata": {664        "id": "dm0If2OWcVGa",665        "colab": {666          "base_uri": "https://localhost:8080/"667        },668        "outputId": "65fe5959-13eb-4280-ba2a-daafa21308fa"669      },670      "execution_count": 4,671      "outputs": [672        {673          "output_type": "execute_result",674          "data": {675            "text/plain": [676              "DatasetDict({\n",677              "    train: Dataset({\n",678              "        features: ['id', 'dialogue', 'summary'],\n",679              "        num_rows: 14732\n",680              "    })\n",681              "    test: Dataset({\n",682              "        features: ['id', 'dialogue', 'summary'],\n",683              "        num_rows: 819\n",684              "    })\n",685              "    validation: Dataset({\n",686              "        features: ['id', 'dialogue', 'summary'],\n",687              "        num_rows: 818\n",688              "    })\n",689              "})"690            ]691          },692          "metadata": {},693          "execution_count": 4694        }695      ]696    },697    {698      "cell_type": "code",699      "source": [700        "#print some of the data\n",701        "\n",702        "split_lengths = [len(dataset_samsum[split])for split in dataset_samsum]\n",703        "\n",704        "print(f\"Split lengths: {split_lengths}\")\n",705        "print(f\"Features: {dataset_samsum['train'].column_names}\")\n",706        "print(\"\\nDialogue:\")\n",707        "\n",708        "print(dataset_samsum[\"test\"][1][\"dialogue\"])\n",709        "\n",710        "print(\"\\nSummary:\")\n",711        "\n",712        "print(dataset_samsum[\"test\"][1][\"summary\"])"713      ],714      "metadata": {715        "id": "PSgRXN_-e1gQ",716        "colab": {717          "base_uri": "https://localhost:8080/"718        },719        "outputId": "e0c607b5-07d8-44b3-dec2-753ac96174af"720      },721      "execution_count": 7,722      "outputs": [723        {724          "output_type": "stream",725          "name": "stdout",726          "text": [727            "Split lengths: [14732, 819, 818]\n",728            "Features: ['id', 'dialogue', 'summary']\n",729            "\n",730            "Dialogue:\n",731            "Eric: MACHINE!\r\n",732            "Rob: That's so gr8!\r\n",733            "Eric: I know! And shows how Americans see Russian ;)\r\n",734            "Rob: And it's really funny!\r\n",735            "Eric: I know! I especially like the train part!\r\n",736            "Rob: Hahaha! No one talks to the machine like that!\r\n",737            "Eric: Is this his only stand-up?\r\n",738            "Rob: Idk. I'll check.\r\n",739            "Eric: Sure.\r\n",740            "Rob: Turns out no! There are some of his stand-ups on youtube.\r\n",741            "Eric: Gr8! I'll watch them now!\r\n",742            "Rob: Me too!\r\n",743            "Eric: MACHINE!\r\n",744            "Rob: MACHINE!\r\n",745            "Eric: TTYL?\r\n",746            "Rob: Sure :)\n",747            "\n",748            "Summary:\n",749            "Eric and Rob are going to watch a stand-up on youtube.\n"750          ]751        }752      ]753    },754    {755      "cell_type": "code",756      "source": [757        "# This Function will help me to convert input IDs to attention mask and levels.\n",758        "def convert_examples_to_features(example_batch):\n",759        "  input_encodings = tokenizer(example_batch['dialogue'], max_length = 1024, truncation = True )\n",760        "\n",761        "  with tokenizer.as_target_tokenizer():\n",762        "     target_encodings = tokenizer(example_batch['summary'], max_length = 128, truncation = True )\n",763        "\n",764        "  return {\n",765        "      'input_ids' : input_encodings['input_ids'],\n",766        "      'attention_mask': input_encodings['attention_mask'],\n",767        "      'labels': target_encodings['input_ids']\n",768        "      }\n"769      ],770      "metadata": {771        "id": "gdvhh5hgfaI2"772      },773      "execution_count": 5,774      "outputs": []775    },776    {777      "cell_type": "code",778      "source": [779        "dataset_samsum_pt = dataset_samsum.map(convert_examples_to_features, batched= True)"780      ],781      "metadata": {782        "id": "tVdWxnKp5z6B"783      },784      "execution_count": 6,785      "outputs": []786    },787    {788      "cell_type": "code",789      "source": [790        "dataset_samsum_pt[\"train\"]"791      ],792      "metadata": {793        "colab": {794          "base_uri": "https://localhost:8080/"795        },796        "id": "5MN7opbt6ZpO",797        "outputId": "27d35d24-9962-4a9b-a3a8-6a179a5f5e87"798      },799      "execution_count": 7,800      "outputs": [801        {802          "output_type": "execute_result",803          "data": {804            "text/plain": [805              "Dataset({\n",806              "    features: ['id', 'dialogue', 'summary', 'input_ids', 'attention_mask', 'labels'],\n",807              "    num_rows: 14732\n",808              "})"809            ]810          },811          "metadata": {},812          "execution_count": 7813        }814      ]815    },816    {817      "cell_type": "code",818      "source": [819        "# Training\n",820        "# This will Creating the baches instead of pass all of the data\n",821        "from transformers import DataCollatorForSeq2Seq\n",822        "\n",823        "seq2seq_data_collator = DataCollatorForSeq2Seq(tokenizer, model = model_pegasus)"824      ],825      "metadata": {826        "id": "gIQ7E1vt6iuY"827      },828      "execution_count": 8,829      "outputs": []830    },831    {832      "cell_type": "code",833      "source": [834        "from transformers import TrainingArguments, Trainer\n",835        "\n",836        "trainer_args = TrainingArguments(\n",837        "    output_dir='pegasus-samsum', num_train_epochs=1, warmup_steps=500,\n",838        "    per_device_train_batch_size=1, per_device_eval_batch_size=1,\n",839        "    weight_decay=0.01, logging_steps=10,\n",840        "    evaluation_strategy='steps', eval_steps=500, save_steps=1e6,\n",841        "    gradient_accumulation_steps=16 )"842      ],843      "metadata": {844        "id": "CxMFRucz7GtR"845      },846      "execution_count": 9,847      "outputs": []848    },849    {850      "cell_type": "code",851      "source": [852        "\n",853        "trainer = Trainer(model=model_pegasus, args=trainer_args,\n",854        "                  tokenizer=tokenizer, data_collator=seq2seq_data_collator,\n",855        "                  train_dataset=dataset_samsum_pt[\"test\"],\n",856        "                  eval_dataset=dataset_samsum_pt[\"validation\"])\n"857      ],858      "metadata": {859        "id": "qGl6jtKi8M3F"860      },861      "execution_count": 10,862      "outputs": []863    },864    {865      "cell_type": "code",866      "source": [867        "trainer.train()"868      ],869      "metadata": {870        "colab": {871          "base_uri": "https://localhost:8080/",872          "height": 75873        },874        "id": "K2t4JZcc88_k",875        "outputId": "ea997fab-ebaa-4f04-a472-00587cfd7bee"876      },877      "execution_count": null,878      "outputs": [879        {880          "output_type": "display_data",881          "data": {882            "text/plain": [883              "<IPython.core.display.HTML object>"884            ],885            "text/html": [886              "\n",887              "    <div>\n",888              "      \n",889              "      <progress value='2' max='51' style='width:300px; height:20px; vertical-align: middle;'></progress>\n",890              "      [ 2/51 : < :, Epoch 0.02/1]\n",891              "    </div>\n",892              "    <table border=\"1\" class=\"dataframe\">\n",893              "  <thead>\n",894              " <tr style=\"text-align: left;\">\n",895              "      <th>Step</th>\n",896              "      <th>Training Loss</th>\n",897              "      <th>Validation Loss</th>\n",898              "    </tr>\n",899              "  </thead>\n",900              "  <tbody>\n",901              "  </tbody>\n",902              "</table><p>"903            ]904          },905          "metadata": {}906        }907      ]908    },909    {910      "cell_type": "code",911      "source": [912        "# Evaluation\n",913        "\n",914        "def generate_batch_sized_chunks(list_of_elements, batch_size):\n",915        "    \"\"\"split the dataset into smaller batches that we can process simultaneously\n",916        "    Yield successive batch-sized chunks from list_of_elements.\"\"\"\n",917        "    for i in range(0, len(list_of_elements), batch_size):\n",918        "        yield list_of_elements[i : i + batch_size]\n",919        "\n",920        "\n",921        "\n",922        "def calculate_metric_on_test_ds(dataset, metric, model, tokenizer,\n",923        "                               batch_size=16, device=device,\n",924        "                               column_text=\"article\",\n",925        "                               column_summary=\"highlights\"):\n",926        "    article_batches = list(generate_batch_sized_chunks(dataset[column_text], batch_size))\n",927        "    target_batches = list(generate_batch_sized_chunks(dataset[column_summary], batch_size))\n",928        "\n",929        "    for article_batch, target_batch in tqdm(\n",930        "        zip(article_batches, target_batches), total=len(article_batches)):\n",931        "\n",932        "        inputs = tokenizer(article_batch, max_length=1024,  truncation=True,\n",933        "                        padding=\"max_length\", return_tensors=\"pt\")\n",934        "\n",935        "        summaries = model.generate(input_ids=inputs[\"input_ids\"].to(device),\n",936        "                         attention_mask=inputs[\"attention_mask\"].to(device),\n",937        "                         length_penalty=0.8, num_beams=8, max_length=128)\n",938        "        ''' parameter for length penalty ensures that the model does not generate sequences that are too long. '''\n",939        "\n",940        "        # Finally, we decode the generated texts,\n",941        "        # replace the  token, and add the decoded texts with the references to the metric.\n",942        "        decoded_summaries = [tokenizer.decode(s, skip_special_tokens=True,\n",943        "                                clean_up_tokenization_spaces=True)\n",944        "               for s in summaries]\n",945        "\n",946        "        decoded_summaries = [d.replace(\"\", \" \") for d in decoded_summaries]\n",947        "\n",948        "\n",949        "        metric.add_batch(predictions=decoded_summaries, references=target_batch)\n",950        "\n",951        "    #  Finally compute and return the ROUGE scores.\n",952        "    score = metric.compute()\n",953        "    return score"954      ],955      "metadata": {956        "id": "if8sc5Q6Czwg"957      },958      "execution_count": 13,959      "outputs": []960    },961    {962      "cell_type": "code",963      "source": [964        "rouge_names = [\"rouge1\", \"rouge2\", \"rougeL\", \"rougeLsum\"]\n",965        "rouge_metric = load_metric('rouge')"966      ],967      "metadata": {968        "colab": {969          "base_uri": "https://localhost:8080/",970          "height": 178,971          "referenced_widgets": [972            "1427097f4c03437ba6545632c57b0027",973            "46fcd96cdb53432a91a36775693d3d1f",974            "124ec5caf31b48b1a890bd6dbeabd51b",975            "a748ec3f25534cedab90d9953d43d85b",976            "cd070b0e2c8b48b3b02848cb89b9cded",977            "f2a57bcaa5d545dc830e6b3d7a6c921e",978            "d442f282871a490b905f2c4562c107db",979            "ae5a78baca7647afb9f095c99e6285ca",980            "4a726f29235948fba545f9bd684c9312",981            "630c7d9853da46aca7cf874d61545e3e",982            "46e64861f65446eab869bb844f53e470"983          ]984        },985        "id": "917gFi2PMG1s",986        "outputId": "294bf00f-80fe-4d58-a6c0-8c2bf1789577"987      },988      "execution_count": 14,989      "outputs": [990        {991          "output_type": "stream",992          "name": "stderr",993          "text": [994            "<ipython-input-14-5a43aadd1b0e>:2: FutureWarning: load_metric is deprecated and will be removed in the next major version of datasets. Use 'evaluate.load' instead, from the new library 🤗 Evaluate: https://huggingface.co/docs/evaluate\n",995            "  rouge_metric = load_metric('rouge')\n",996            "/usr/local/lib/python3.10/dist-packages/datasets/load.py:752: FutureWarning: The repository for rouge contains custom code which must be executed to correctly load the metric. You can inspect the repository content at https://raw.githubusercontent.com/huggingface/datasets/2.16.1/metrics/rouge/rouge.py\n",997            "You can avoid this message in future by passing the argument `trust_remote_code=True`.\n",998            "Passing `trust_remote_code=True` will be mandatory to load this metric from the next major release of `datasets`.\n",999            "  warnings.warn(\n"1000          ]1001        },1002        {1003          "output_type": "display_data",1004          "data": {1005            "text/plain": [1006              "Downloading builder script:   0%|          | 0.00/2.17k [00:00<?, ?B/s]"1007            ],1008            "application/vnd.jupyter.widget-view+json": {1009              "version_major": 2,1010              "version_minor": 0,1011              "model_id": "1427097f4c03437ba6545632c57b0027"1012            }1013          },1014          "metadata": {}1015        }1016      ]1017    },1018    {1019      "cell_type": "code",1020      "source": [1021        "score = calculate_metric_on_test_ds(\n",1022        "    dataset_samsum['test'][0:10], rouge_metric, trainer.model, tokenizer, batch_size = 2, column_text = 'dialogue', column_summary= 'summary'\n",1023        ")\n",1024        "\n",1025        "rouge_dict = dict((rn, score[rn].mid.fmeasure ) for rn in rouge_names )\n",1026        "\n",1027        "pd.DataFrame(rouge_dict, index = [f'pegasus'] )"1028      ],1029      "metadata": {1030        "colab": {1031          "base_uri": "https://localhost:8080/",1032          "height": 991033        },1034        "id": "z1BOJpvuMKNL",1035        "outputId": "011e6644-f474-4f89-fed9-170fabf62f9e"1036      },1037      "execution_count": 15,1038      "outputs": [1039        {1040          "output_type": "stream",1041          "name": "stderr",1042          "text": [1043            "100%|██████████| 5/5 [07:01<00:00, 84.37s/it]\n"1044          ]1045        },1046        {1047          "output_type": "execute_result",1048          "data": {1049            "text/plain": [1050              "           rouge1  rouge2    rougeL  rougeLsum\n",1051              "pegasus  0.018027     0.0  0.017991   0.017965"1052            ],1053            "text/html": [1054              "\n",1055              "  <div id=\"df-4f8ee198-2039-47f1-9df5-110a70a6b883\" class=\"colab-df-container\">\n",1056              "    <div>\n",1057              "<style scoped>\n",1058              "    .dataframe tbody tr th:only-of-type {\n",1059              "        vertical-align: middle;\n",1060              "    }\n",1061              "\n",1062              "    .dataframe tbody tr th {\n",1063              "        vertical-align: top;\n",1064              "    }\n",1065              "\n",1066              "    .dataframe thead th {\n",1067              "        text-align: right;\n",1068              "    }\n",1069              "</style>\n",1070              "<table border=\"1\" class=\"dataframe\">\n",1071              "  <thead>\n",1072              "    <tr style=\"text-align: right;\">\n",1073              "      <th></th>\n",1074              "      <th>rouge1</th>\n",1075              "      <th>rouge2</th>\n",1076              "      <th>rougeL</th>\n",1077              "      <th>rougeLsum</th>\n",1078              "    </tr>\n",1079              "  </thead>\n",1080              "  <tbody>\n",1081              "    <tr>\n",1082              "      <th>pegasus</th>\n",1083              "      <td>0.018027</td>\n",1084              "      <td>0.0</td>\n",1085              "      <td>0.017991</td>\n",1086              "      <td>0.017965</td>\n",1087              "    </tr>\n",1088              "  </tbody>\n",1089              "</table>\n",1090              "</div>\n",1091              "    <div class=\"colab-df-buttons\">\n",1092              "\n",1093              "  <div class=\"colab-df-container\">\n",1094              "    <button class=\"colab-df-convert\" onclick=\"convertToInteractive('df-4f8ee198-2039-47f1-9df5-110a70a6b883')\"\n",1095              "            title=\"Convert this dataframe to an interactive table.\"\n",1096              "            style=\"display:none;\">\n",1097              "\n",1098              "  <svg xmlns=\"http://www.w3.org/2000/svg\" height=\"24px\" viewBox=\"0 -960 960 960\">\n",1099              "    <path d=\"M120-120v-720h720v720H120Zm60-500h600v-160H180v160Zm220 220h160v-160H400v160Zm0 220h160v-160H400v160ZM180-400h160v-160H180v160Zm440 0h160v-160H620v160ZM180-180h160v-160H180v160Zm440 0h160v-160H620v160Z\"/>\n",1100              "  </svg>\n",1101              "    </button>\n",1102              "\n",1103              "  <style>\n",1104              "    .colab-df-container {\n",1105              "      display:flex;\n",1106              "      gap: 12px;\n",1107              "    }\n",1108              "\n",1109              "    .colab-df-convert {\n",1110              "      background-color: #E8F0FE;\n",1111              "      border: none;\n",1112              "      border-radius: 50%;\n",1113              "      cursor: pointer;\n",1114              "      display: none;\n",1115              "      fill: #1967D2;\n",1116              "      height: 32px;\n",1117              "      padding: 0 0 0 0;\n",1118              "      width: 32px;\n",1119              "    }\n",1120              "\n",1121              "    .colab-df-convert:hover {\n",1122              "      background-color: #E2EBFA;\n",1123              "      box-shadow: 0px 1px 2px rgba(60, 64, 67, 0.3), 0px 1px 3px 1px rgba(60, 64, 67, 0.15);\n",1124              "      fill: #174EA6;\n",1125              "    }\n",1126              "\n",1127              "    .colab-df-buttons div {\n",1128              "      margin-bottom: 4px;\n",1129              "    }\n",1130              "\n",1131              "    [theme=dark] .colab-df-convert {\n",1132              "      background-color: #3B4455;\n",1133              "      fill: #D2E3FC;\n",1134              "    }\n",1135              "\n",1136              "    [theme=dark] .colab-df-convert:hover {\n",1137              "      background-color: #434B5C;\n",1138              "      box-shadow: 0px 1px 3px 1px rgba(0, 0, 0, 0.15);\n",1139              "      filter: drop-shadow(0px 1px 2px rgba(0, 0, 0, 0.3));\n",1140              "      fill: #FFFFFF;\n",1141              "    }\n",1142              "  </style>\n",1143              "\n",1144              "    <script>\n",1145              "      const buttonEl =\n",1146              "        document.querySelector('#df-4f8ee198-2039-47f1-9df5-110a70a6b883 button.colab-df-convert');\n",1147              "      buttonEl.style.display =\n",1148              "        google.colab.kernel.accessAllowed ? 'block' : 'none';\n",1149              "\n",1150              "      async function convertToInteractive(key) {\n",1151              "        const element = document.querySelector('#df-4f8ee198-2039-47f1-9df5-110a70a6b883');\n",1152              "        const dataTable =\n",1153              "          await google.colab.kernel.invokeFunction('convertToInteractive',\n",1154              "                                                    [key], {});\n",1155              "        if (!dataTable) return;\n",1156              "\n",1157              "        const docLinkHtml = 'Like what you see? Visit the ' +\n",1158              "          '<a target=\"_blank\" href=https://colab.research.google.com/notebooks/data_table.ipynb>data table notebook</a>'\n",1159              "          + ' to learn more about interactive tables.';\n",1160              "        element.innerHTML = '';\n",1161              "        dataTable['output_type'] = 'display_data';\n",1162              "        await google.colab.output.renderOutput(dataTable, element);\n",1163              "        const docLink = document.createElement('div');\n",1164              "        docLink.innerHTML = docLinkHtml;\n",1165              "        element.appendChild(docLink);\n",1166              "      }\n",1167              "    </script>\n",1168              "  </div>\n",1169              "\n",1170              "\n",1171              "    </div>\n",1172              "  </div>\n"1173            ]1174          },1175          "metadata": {},1176          "execution_count": 151177        }1178      ]1179    },1180    {1181      "cell_type": "code",1182      "source": [1183        "## Save model\n",1184        "model_pegasus.save_pretrained(\"pegasus-samsum-model\")"1185      ],1186      "metadata": {1187        "id": "8g-0JoLnMRbI"1188      },1189      "execution_count": null,1190      "outputs": []1191    },1192    {1193      "cell_type": "code",1194      "source": [1195        "## Save tokenizer\n",1196        "tokenizer.save_pretrained(\"tokenizer\")"1197      ],1198      "metadata": {1199        "id": "AaX9TwCpS0uf"1200      },

Showing the first 1,200 of 1258 lines. Download the file for the rest.