{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-06T10:47:33.196201Z","iopub.execute_input":"2023-08-06T10:47:33.196654Z","iopub.status.idle":"2023-08-06T10:47:33.231515Z","shell.execute_reply.started":"2023-08-06T10:47:33.196613Z","shell.execute_reply":"2023-08-06T10:47:33.230628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\", nrows=500)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-06T10:47:33.233158Z","iopub.execute_input":"2023-08-06T10:47:33.234127Z","iopub.status.idle":"2023-08-06T10:47:33.282864Z","shell.execute_reply.started":"2023-08-06T10:47:33.234084Z","shell.execute_reply":"2023-08-06T10:47:33.281779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transformers","metadata":{"execution":{"iopub.status.busy":"2023-08-06T10:47:33.284488Z","iopub.execute_input":"2023-08-06T10:47:33.285395Z","iopub.status.idle":"2023-08-06T10:47:48.923003Z","shell.execute_reply.started":"2023-08-06T10:47:33.285360Z","shell.execute_reply":"2023-08-06T10:47:48.921587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer\n\ntokenizer = AutoTokenizer.from_pretrained(\"bert-base-uncased\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T10:47:48.924640Z","iopub.execute_input":"2023-08-06T10:47:48.925035Z","iopub.status.idle":"2023-08-06T10:47:53.638026Z","shell.execute_reply.started":"2023-08-06T10:47:48.925001Z","shell.execute_reply":"2023-08-06T10:47:53.636777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install datasets","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:09:06.858882Z","iopub.execute_input":"2023-08-06T11:09:06.859346Z","iopub.status.idle":"2023-08-06T11:09:20.597795Z","shell.execute_reply.started":"2023-08-06T11:09:06.859313Z","shell.execute_reply":"2023-08-06T11:09:20.596345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import Dataset\n\ntrain_ds = Dataset.from_pandas(data)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:10:58.642873Z","iopub.execute_input":"2023-08-06T11:10:58.643295Z","iopub.status.idle":"2023-08-06T11:10:59.242770Z","shell.execute_reply.started":"2023-08-06T11:10:58.643258Z","shell.execute_reply":"2023-08-06T11:10:59.241854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"type(train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:11:07.958981Z","iopub.execute_input":"2023-08-06T11:11:07.959845Z","iopub.status.idle":"2023-08-06T11:11:07.968320Z","shell.execute_reply.started":"2023-08-06T11:11:07.959800Z","shell.execute_reply":"2023-08-06T11:11:07.967000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:19:48.574907Z","iopub.execute_input":"2023-08-06T11:19:48.575341Z","iopub.status.idle":"2023-08-06T11:19:48.583859Z","shell.execute_reply.started":"2023-08-06T11:19:48.575308Z","shell.execute_reply":"2023-08-06T11:19:48.582266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(dataset):\n    return tokenizer(dataset[\"question_text\"],return_tensors=\"np\",padding=True)\n\ntokenized_dataset = preprocess(train_ds)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:21:01.392450Z","iopub.execute_input":"2023-08-06T11:21:01.392916Z","iopub.status.idle":"2023-08-06T11:21:01.458265Z","shell.execute_reply.started":"2023-08-06T11:21:01.392881Z","shell.execute_reply":"2023-08-06T11:21:01.456958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = np.array(train_ds[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:21:04.158502Z","iopub.execute_input":"2023-08-06T11:21:04.159031Z","iopub.status.idle":"2023-08-06T11:21:04.166527Z","shell.execute_reply.started":"2023-08-06T11:21:04.158990Z","shell.execute_reply":"2023-08-06T11:21:04.164911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenized_dataset = dict(tokenized_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:21:06.201152Z","iopub.execute_input":"2023-08-06T11:21:06.201687Z","iopub.status.idle":"2023-08-06T11:21:06.211344Z","shell.execute_reply.started":"2023-08-06T11:21:06.201647Z","shell.execute_reply":"2023-08-06T11:21:06.209045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TFAutoModelForSequenceClassification\n\nmodel = TFAutoModelForSequenceClassification.from_pretrained(\"bert-base-uncased\")\n\nmodel.compile(optimizer=\"adam\")","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:16:31.301269Z","iopub.execute_input":"2023-08-06T11:16:31.301736Z","iopub.status.idle":"2023-08-06T11:16:42.025791Z","shell.execute_reply.started":"2023-08-06T11:16:31.301697Z","shell.execute_reply":"2023-08-06T11:16:42.024485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(tokenized_dataset,labels)","metadata":{"execution":{"iopub.status.busy":"2023-08-06T11:21:10.745987Z","iopub.execute_input":"2023-08-06T11:21:10.747388Z","iopub.status.idle":"2023-08-06T11:26:13.158457Z","shell.execute_reply.started":"2023-08-06T11:21:10.747310Z","shell.execute_reply":"2023-08-06T11:26:13.157388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}