{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-06T18:02:38.601891Z","iopub.execute_input":"2023-08-06T18:02:38.602455Z","iopub.status.idle":"2023-08-06T18:02:38.616917Z","shell.execute_reply.started":"2023-08-06T18:02:38.602414Z","shell.execute_reply":"2023-08-06T18:02:38.615516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\", nrows=500)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:02:38.625627Z","iopub.execute_input":"2023-08-06T18:02:38.626557Z","iopub.status.idle":"2023-08-06T18:02:38.652148Z","shell.execute_reply.started":"2023-08-06T18:02:38.626526Z","shell.execute_reply":"2023-08-06T18:02:38.650909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:15:56.077096Z","iopub.execute_input":"2023-08-06T18:15:56.079348Z","iopub.status.idle":"2023-08-06T18:15:56.091719Z","shell.execute_reply.started":"2023-08-06T18:15:56.079299Z","shell.execute_reply":"2023-08-06T18:15:56.090853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transformers accelerate bitsandbytes peft","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:02:38.653762Z","iopub.execute_input":"2023-08-06T18:02:38.654250Z","iopub.status.idle":"2023-08-06T18:02:50.627039Z","shell.execute_reply.started":"2023-08-06T18:02:38.654211Z","shell.execute_reply":"2023-08-06T18:02:50.625827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install datasets","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:02:50.631165Z","iopub.execute_input":"2023-08-06T18:02:50.631579Z","iopub.status.idle":"2023-08-06T18:03:03.140987Z","shell.execute_reply.started":"2023-08-06T18:02:50.631544Z","shell.execute_reply":"2023-08-06T18:03:03.139669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import Dataset\n\ntrain_ds = Dataset.from_pandas(data[:400])\ntest_ds = Dataset.from_pandas(data[400:])","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:03.142912Z","iopub.execute_input":"2023-08-06T18:03:03.143354Z","iopub.status.idle":"2023-08-06T18:03:03.796891Z","shell.execute_reply.started":"2023-08-06T18:03:03.143285Z","shell.execute_reply":"2023-08-06T18:03:03.795856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer, AutoModelForSeq2SeqLM\n\ntokenizer = AutoTokenizer.from_pretrained(\"google/flan-t5-small\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:03.798488Z","iopub.execute_input":"2023-08-06T18:03:03.799069Z","iopub.status.idle":"2023-08-06T18:03:04.636762Z","shell.execute_reply.started":"2023-08-06T18:03:03.799033Z","shell.execute_reply":"2023-08-06T18:03:04.635538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSeq2SeqLM.from_pretrained(\"google/flan-t5-small\",load_in_8bit=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:04.638285Z","iopub.execute_input":"2023-08-06T18:03:04.638938Z","iopub.status.idle":"2023-08-06T18:03:13.840683Z","shell.execute_reply.started":"2023-08-06T18:03:04.638909Z","shell.execute_reply":"2023-08-06T18:03:13.839666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from peft import prepare_model_for_int8_training\n\nmodel = prepare_model_for_int8_training(model)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:13.842393Z","iopub.execute_input":"2023-08-06T18:03:13.843140Z","iopub.status.idle":"2023-08-06T18:03:13.874053Z","shell.execute_reply.started":"2023-08-06T18:03:13.843103Z","shell.execute_reply":"2023-08-06T18:03:13.872304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from peft import LoraConfig, get_peft_model\n\nloraconfig =LoraConfig(r=16, lora_alpha=32, target_modules=['q','v'], lora_dropout=0.05, bias=\"none\", task_type=\"SEQ_2_SEQ_LM\")\n\nmodel = get_peft_model(model,loraconfig)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:13.875817Z","iopub.execute_input":"2023-08-06T18:03:13.876567Z","iopub.status.idle":"2023-08-06T18:03:14.019138Z","shell.execute_reply.started":"2023-08-06T18:03:13.876524Z","shell.execute_reply":"2023-08-06T18:03:14.018095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize_function(examples):\n    return tokenizer(examples[\"question_text\"], padding=\"max_length\", truncation=True)\n\n\n#train_tokenized_datasets = train_ds.map(tokenize_function, batched=True)\n#test_tokenized_datasets = test_ds.map(tokenize_function, batched=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:14.020782Z","iopub.execute_input":"2023-08-06T18:03:14.021157Z","iopub.status.idle":"2023-08-06T18:03:14.272717Z","shell.execute_reply.started":"2023-08-06T18:03:14.021124Z","shell.execute_reply":"2023-08-06T18:03:14.271457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.columns.tolist()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:14.274748Z","iopub.execute_input":"2023-08-06T18:03:14.275169Z","iopub.status.idle":"2023-08-06T18:03:14.283033Z","shell.execute_reply.started":"2023-08-06T18:03:14.275131Z","shell.execute_reply":"2023-08-06T18:03:14.281765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_dict = {1:'insincere',0:'sincere'}","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:24:24.862552Z","iopub.execute_input":"2023-08-06T18:24:24.862941Z","iopub.status.idle":"2023-08-06T18:24:24.867859Z","shell.execute_reply.started":"2023-08-06T18:24:24.862911Z","shell.execute_reply":"2023-08-06T18:24:24.866865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:45:35.252683Z","iopub.execute_input":"2023-08-06T18:45:35.253731Z","iopub.status.idle":"2023-08-06T18:45:35.261435Z","shell.execute_reply.started":"2023-08-06T18:45:35.253693Z","shell.execute_reply":"2023-08-06T18:45:35.260195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data preprocessing\ntext_column = \"question_text\"\nlabel_column = \"target\"\nmax_length = 128\n\n\ndef preprocess_function(examples):\n    inputs = examples[text_column]\n    targets = [str(x) for x in examples[label_column]] \n    # Don't know why but converting 1 to '1' and then encoding gives better results than converting 1 to 'insincere'!\n    # May be because the sampling is poor and just 500 records are selected at random\n    #print(type(targets[2]))\n    #print(type(inputs[2]))\n    model_inputs = tokenizer(inputs, max_length=max_length, padding=\"max_length\", truncation=True, return_tensors=\"np\")\n    #print(model_inputs)\n    labels = tokenizer(targets, max_length=3, padding=\"max_length\", truncation=True, return_tensors=\"np\")\n    #print(labels)\n    labels = labels[\"input_ids\"]\n    labels[labels == tokenizer.pad_token_id] = -100\n    model_inputs[\"labels\"] = labels#.tolist()\n    #print(model_inputs[\"labels\"])\n    #labels1 = np.array(examples[\"target\"]).reshape(-1,1)\n    #model_inputs[\"labels\"] = labels1.tolist()\n    return model_inputs\n\n\ntrain_dataset = train_ds.map(\n    preprocess_function,\n    batched=True,\n    num_proc=1,\n    remove_columns=data.columns.tolist(),\n    load_from_cache_file=False,\n    desc=\"Running tokenizer on dataset\",\n)\n\ntest_dataset = test_ds.map(\n    preprocess_function,\n    batched=True,\n    num_proc=1,\n    remove_columns=data.columns.tolist(),\n    load_from_cache_file=False,\n    desc=\"Running tokenizer on dataset\",\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:57:07.347252Z","iopub.execute_input":"2023-08-06T18:57:07.348231Z","iopub.status.idle":"2023-08-06T18:57:07.483675Z","shell.execute_reply.started":"2023-08-06T18:57:07.348193Z","shell.execute_reply":"2023-08-06T18:57:07.482709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\ntrain_tokenized_datasets = train_tokenized_datasets.remove_columns(['qid','question_text'])\ntrain_tokenized_datasets = train_tokenized_datasets.rename_column('target','labels')\ntrain_tokenized_datasets = train_tokenized_datasets.with_format(\"torch\")\n\ntest_tokenized_datasets = test_tokenized_datasets.remove_columns(['qid','question_text'])\ntest_tokenized_datasets = test_tokenized_datasets.rename_column('target','labels')\ntest_tokenized_datasets = test_tokenized_datasets.with_format(\"torch\")\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:14.436554Z","iopub.execute_input":"2023-08-06T18:03:14.437580Z","iopub.status.idle":"2023-08-06T18:03:14.446270Z","shell.execute_reply.started":"2023-08-06T18:03:14.437538Z","shell.execute_reply":"2023-08-06T18:03:14.444779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_tokenized_datasets","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:03:14.448564Z","iopub.execute_input":"2023-08-06T18:03:14.449550Z","iopub.status.idle":"2023-08-06T18:03:14.457492Z","shell.execute_reply.started":"2023-08-06T18:03:14.449483Z","shell.execute_reply":"2023-08-06T18:03:14.455861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:42:08.811466Z","iopub.execute_input":"2023-08-06T18:42:08.812584Z","iopub.status.idle":"2023-08-06T18:42:08.819909Z","shell.execute_reply.started":"2023-08-06T18:42:08.812537Z","shell.execute_reply":"2023-08-06T18:42:08.818775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[100]['labels']","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:57:59.194468Z","iopub.execute_input":"2023-08-06T18:57:59.194905Z","iopub.status.idle":"2023-08-06T18:57:59.207169Z","shell.execute_reply.started":"2023-08-06T18:57:59.194872Z","shell.execute_reply":"2023-08-06T18:57:59.205851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.print_trainable_parameters()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:22:46.885969Z","iopub.execute_input":"2023-08-06T18:22:46.886412Z","iopub.status.idle":"2023-08-06T18:22:46.897229Z","shell.execute_reply.started":"2023-08-06T18:22:46.886371Z","shell.execute_reply":"2023-08-06T18:22:46.895876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import DataCollatorForSeq2Seq, DataCollatorWithPadding\n\n#data_collator = DataCollatorForSeq2Seq(tokenizer,model=model,pad_to_multiple_of=8,label_pad_token_id=-100)\n\ndata_collator = DataCollatorWithPadding(tokenizer,max_length =128,return_tensors =\"pt\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:04:37.159207Z","iopub.execute_input":"2023-08-06T18:04:37.159724Z","iopub.status.idle":"2023-08-06T18:04:37.231263Z","shell.execute_reply.started":"2023-08-06T18:04:37.159692Z","shell.execute_reply":"2023-08-06T18:04:37.230342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TrainingArguments, Trainer\n\ntraining_args = TrainingArguments(\n    \"temp\",\n    evaluation_strategy=\"epoch\",\n    learning_rate=1e-3,\n    gradient_accumulation_steps=1,\n    auto_find_batch_size=True,\n    num_train_epochs=1,\n    save_steps=100,\n    save_total_limit=8,\n    report_to=\"none\"\n)\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    #data_collator = data_collator,\n    train_dataset=train_dataset,\n    eval_dataset=test_dataset,\n)\nmodel.config.use_cache = False","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:57:30.960606Z","iopub.execute_input":"2023-08-06T18:57:30.961635Z","iopub.status.idle":"2023-08-06T18:57:30.974969Z","shell.execute_reply.started":"2023-08-06T18:57:30.961597Z","shell.execute_reply":"2023-08-06T18:57:30.973710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T18:57:34.996747Z","iopub.execute_input":"2023-08-06T18:57:34.997136Z","iopub.status.idle":"2023-08-06T18:57:47.873994Z","shell.execute_reply.started":"2023-08-06T18:57:34.997105Z","shell.execute_reply":"2023-08-06T18:57:47.873003Z"},"trusted":true},"execution_count":null,"outputs":[]}]}