{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":7615777,"sourceType":"datasetVersion","datasetId":4435237}],"dockerImageVersionId":30648,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-17T15:47:56.975874Z","iopub.execute_input":"2024-02-17T15:47:56.976563Z","iopub.status.idle":"2024-02-17T15:47:57.341533Z","shell.execute_reply.started":"2024-02-17T15:47:56.976526Z","shell.execute_reply":"2024-02-17T15:47:57.340630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom transformers import BartTokenizer, BartForConditionalGeneration, Trainer, TrainingArguments\nfrom torch.utils.data import Dataset\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:47:59.781284Z","iopub.execute_input":"2024-02-17T15:47:59.782334Z","iopub.status.idle":"2024-02-17T15:48:20.362643Z","shell.execute_reply.started":"2024-02-17T15:47:59.782294Z","shell.execute_reply":"2024-02-17T15:48:20.361879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/wikitest-eng0/english-0.tsv'\ni=0\n# Load your dataset\ndf = pd.read_csv(file_path, sep='\\t', header=None,skiprows=1, nrows=50000)  # Assuming no header\ndel df[df.columns[0]]\ndf.fillna('', inplace=True)\ntexts = df.iloc[:, 15].tolist()  # Texts are in column 16 (0-indexed)\nsummaries = df.iloc[:, 17].tolist()  # Summaries/Captions are in column 18\nfor text in texts:\n    print(text)\n    print(\"--------------------------\")\n    i=i+1;\n    if(i==5):\n        break\ni=0\nfor text in summaries:\n    print(text)\n    print(\"--------------------------\")\n    i=i+1;\n    if(i==5):\n        break\ntexts_val = list(texts[-10000:])\nsummaries_val = list(summaries[-10000:])\ntexts = list(texts[:-10000])\nsummaries = list(summaries[:-10000])\n","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:48:20.364572Z","iopub.execute_input":"2024-02-17T15:48:20.365321Z","iopub.status.idle":"2024-02-17T15:48:20.957846Z","shell.execute_reply.started":"2024-02-17T15:48:20.365291Z","shell.execute_reply":"2024-02-17T15:48:20.956893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install wandb\nimport wandb\nwandb.login(key=\"b94821f49408c3804f54372ec4f3cdb17c319aec\")\n# Define a custom dataset for tokenizing texts and summaries\nclass SummaryDataset(Dataset):\n    def __init__(self, tokenizer, texts, summaries, max_length=512, max_summary_length=128):\n        self.tokenizer = tokenizer\n        self.texts = texts\n        self.summaries = summaries\n        self.max_length = max_length\n        self.max_summary_length = max_summary_length\n        \n    def __len__(self):\n        return len(self.texts)\n    \n    def __getitem__(self, index):\n        text = self.texts[index]\n        summary = self.summaries[index]\n        \n        inputs = self.tokenizer(text, max_length=self.max_length, padding='max_length', truncation=True, return_tensors=\"pt\")\n        outputs = self.tokenizer(summary, max_length=self.max_summary_length, padding='max_length', truncation=True, return_tensors=\"pt\")\n        \n#         print(len(text), len(summary))\n#         print(inputs.input_ids.squeeze().shape, inputs.attention_mask.squeeze().shape, outputs.input_ids.squeeze().shape)\n        return {\n            \"input_ids\": inputs.input_ids.squeeze(),\n            \"attention_mask\": inputs.attention_mask.squeeze(),\n            \"labels\": outputs.input_ids.squeeze()\n        }\n\n# Initialize tokenizer and model\nmodel_name = 'facebook/bart-large-cnn'\ntokenizer = BartTokenizer.from_pretrained(model_name)\nmodel = BartForConditionalGeneration.from_pretrained(model_name)\n\n# Prepare dataset\ndataset = SummaryDataset(tokenizer, texts, summaries)\nvalidation_dataset = SummaryDataset(tokenizer, texts_val, summaries_val)\n\n# Define training arguments\ntraining_args = TrainingArguments(\n    output_dir='./results',\n    num_train_epochs=5,\n    per_device_train_batch_size=4,\n    per_device_eval_batch_size=4,\n    warmup_steps=500,\n    weight_decay=0.01,\n    logging_dir='./logs',\n    logging_steps=10,\n    save_steps=2000, \n    save_total_limit=3,\n    evaluation_strategy=\"epoch\",\n    learning_rate=1e-4\n)\n\n# Initialize Trainer\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=dataset,\n    eval_dataset=validation_dataset, # Add a validation dataset if available\n)\n\n# Train the model\ntrainer.train()\n\n# Save the model\nmodel.save_pretrained('./bart-finetuned')\ntokenizer.save_pretrained('./tokenizer-finetuned')\n","metadata":{"execution":{"iopub.status.busy":"2024-02-17T15:48:20.958962Z","iopub.execute_input":"2024-02-17T15:48:20.959242Z","iopub.status.idle":"2024-02-17T17:17:56.837453Z","shell.execute_reply.started":"2024-02-17T15:48:20.959217Z","shell.execute_reply":"2024-02-17T17:17:56.836418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import transformers\n# from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, Seq2SeqTrainingArguments, Seq2SeqTrainer\n# from datasets import load_dataset, load_from_disk\n# import numpy as np\n# import nltk\n# nltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2024-02-14T04:50:22.932840Z","iopub.status.idle":"2024-02-14T04:50:22.933241Z","shell.execute_reply.started":"2024-02-14T04:50:22.933053Z","shell.execute_reply":"2024-02-14T04:50:22.933070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# max_input = 512\n# max_target = 128\n# batch_size = 3\n# model_checkpoints = \"facebook/bart-large-xsum\"","metadata":{"execution":{"iopub.status.busy":"2024-02-14T04:50:22.934916Z","iopub.status.idle":"2024-02-14T04:50:22.935391Z","shell.execute_reply.started":"2024-02-14T04:50:22.935161Z","shell.execute_reply":"2024-02-14T04:50:22.935181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tokenizer = AutoTokenizer.from_pretrained(model_checkpoints)","metadata":{"execution":{"iopub.status.busy":"2024-02-14T04:50:22.936741Z","iopub.status.idle":"2024-02-14T04:50:22.937267Z","shell.execute_reply.started":"2024-02-14T04:50:22.936996Z","shell.execute_reply":"2024-02-14T04:50:22.937025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def preprocess_data(data_to_process):\n#   #get all the dialogues\n#   #tokenize the dialogues\n#   model_inputs = tokenizer(text,  max_length=max_input, padding='max_length', truncation=True)\n#   #tokenize the summaries\n#   with tokenizer.as_target_tokenizer():\n#     targets = tokenizer(data_to_process['summary'], max_length=max_target, padding='max_length', truncation=True)\n    \n#   #set labels\n#   model_inputs['labels'] = targets['input_ids']\n#   #return the tokenized data\n#   #input_ids, attention_mask and labels\n#   return model_inputs","metadata":{"execution":{"iopub.status.busy":"2024-02-14T04:50:22.938840Z","iopub.status.idle":"2024-02-14T04:50:22.939314Z","shell.execute_reply.started":"2024-02-14T04:50:22.939076Z","shell.execute_reply":"2024-02-14T04:50:22.939094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data=\n# tokenize_data = data.map(preprocess_data, batched = True)","metadata":{"execution":{"iopub.status.busy":"2024-02-08T19:25:29.399865Z","iopub.execute_input":"2024-02-08T19:25:29.40061Z","iopub.status.idle":"2024-02-08T19:25:29.589169Z","shell.execute_reply.started":"2024-02-08T19:25:29.400577Z","shell.execute_reply":"2024-02-08T19:25:29.587425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class DataTransform(Transform):\n#     def __init__(self, tokenizer:PreTrainedTokenizer, column:str):\n#         self.tokenizer = tokenizer\n#         self.column = column\n        \n#     def encodes(self, inp):  \n#         tokenized = self.tokenizer.batch_encode_plus(\n#             [list(inp[self.column])],\n#             max_length=args.max_seq_len, \n#             pad_to_max_length=True, \n#             return_tensors='pt'\n#         )\n#         return TensorText(tokenized['input_ids']).squeeze()\n        \n#     def decodes(self, encoded):\n#         decoded = [\n#             self.tokenizer.decode(\n#                 o, \n#                 skip_special_tokens=True, \n#                 clean_up_tokenization_spaces=False\n#             ) for o in encoded\n#         ]\n#         return decoded\n","metadata":{"execution":{"iopub.status.busy":"2024-02-08T19:35:48.809371Z","iopub.execute_input":"2024-02-08T19:35:48.810089Z","iopub.status.idle":"2024-02-08T19:35:48.868074Z","shell.execute_reply.started":"2024-02-08T19:35:48.810054Z","shell.execute_reply":"2024-02-08T19:35:48.866761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %reload_ext autoreload\n# %autoreload 2\n# !pip install fastai2\n","metadata":{"execution":{"iopub.status.busy":"2024-02-08T19:40:59.217499Z","iopub.execute_input":"2024-02-08T19:40:59.218475Z","iopub.status.idle":"2024-02-08T19:41:12.411408Z","shell.execute_reply.started":"2024-02-08T19:40:59.218433Z","shell.execute_reply":"2024-02-08T19:41:12.410113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import sys\n# import logging\n# logging.getLogger().setLevel(100)\n# from fastprogress import progress_bar\n# from fastai2. import Transform\n","metadata":{"execution":{"iopub.status.busy":"2024-02-08T19:42:24.208481Z","iopub.execute_input":"2024-02-08T19:42:24.209512Z","iopub.status.idle":"2024-02-08T19:42:24.621342Z","shell.execute_reply.started":"2024-02-08T19:42:24.209474Z","shell.execute_reply":"2024-02-08T19:42:24.620061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# import torch\n# from sklearn.model_selection import train_test_split\n# from fastai.data.transforms import RandomSplitter\n# from fastai.data.core import Datasets\n# # Assuming DataTransform is defined elsewhere or you have a similar function ready\n\n# class Namespace:\n#     def __init__(self, **kwargs):\n#         self.__dict__.update(kwargs)\n\n# args = Namespace(\n#     batch_size=4,\n#     max_seq_len=512,\n#     data_path=file_path,  # Update this path to your actual TSV file\n#     device=torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n#     subset=None,\n#     test_pct=0.1\n# )\n\n# # Load dataset from a TSV file, select columns 17 and 18, replace NaN with empty strings\n# df = pd.read_csv(file_path, sep='\\t', header=None, skiprows=1, usecols=[16, 17])\n# df.fillna('', inplace=True)\n# df.columns = ['text', 'summary']  # Temporarily naming columns for easier reference\n\n# # If you want to work with a subset, adjust here; for example, df = df.iloc[:100] for the first 100 rows\n# if args.subset is not None:\n#     df = df.iloc[:100]\n\n# # Filter out rows where the summary is an empty string\n# df = df[df['summary'] != '']\n\n# # Split the dataset\n# train_ds, test_ds = train_test_split(df, test_size=args.test_pct, random_state=42)\n# valid_ds, test_ds = train_test_split(test_ds, test_size=0.5, random_state=42)\n# tokenizer = BartTokenizer.from_pretrained('facebook/bart-large-cnn', add_prefix_space=True)\n# class DataTransform(Transform):\n#     def __init__(self, tokenizer:PreTrainedTokenizer, column:str):\n#         self.tokenizer = tokenizer\n#         self.column = column\n        \n#     def encodes(self, inp):  \n#         tokenized = self.tokenizer.batch_encode_plus(\n#             [list(inp[self.column])],\n#             max_length=args.max_seq_len, \n#             pad_to_max_length=True, \n#             return_tensors='pt'\n#         )\n#         return TensorText(tokenized['input_ids']).squeeze()\n        \n#     def decodes(self, encoded):\n#         decoded = [\n#             self.tokenizer.decode(\n#                 o, \n#                 skip_special_tokens=True, \n#                 clean_up_tokenization_spaces=False\n#             ) for o in encoded\n#         ]\n#         return decoded\n# # Define transformations - assuming DataTransform can handle DataFrame rows directly or is adjusted accordingly\n# x_tfms = [lambda x: DataTransform(tokenizer, text=x['text'])]  # Example transformation, adjust as necessary\n# y_tfms = [lambda x: DataTransform(tokenizer, text=x['summary'])]  # Adjust as necessary\n\n# # Create Datasets object with the transformations\n# dss = Datasets(\n#     train_ds.to_dict('records'),  # Convert DataFrame to a list of dictionaries\n#     tfms=[x_tfms, y_tfms],\n#     splits=RandomSplitter(valid_pct=0.1)(range(len(train_ds)))\n# )\n\n# # Ensure DataTransform and any other custom code is compatible with the input format and adjust accordingly\n","metadata":{"execution":{"iopub.status.busy":"2024-02-08T19:38:36.7985Z","iopub.execute_input":"2024-02-08T19:38:36.799379Z","iopub.status.idle":"2024-02-08T19:39:01.402517Z","shell.execute_reply.started":"2024-02-08T19:38:36.799345Z","shell.execute_reply":"2024-02-08T19:39:01.401108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}