{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30839,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Library","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom transformers import AutoTokenizer, AutoModelForSequenceClassification\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom transformers import AutoTokenizer, DataCollatorWithPadding, TrainingArguments, Trainer\nfrom sklearn.model_selection import KFold\nfrom datasets import Dataset\nfrom sklearn.metrics import accuracy_score, precision_recall_fscore_support\nimport random\nimport torch\n\n# Set seed for reproducibility\ndef set_seed(seed=42):\n    \"\"\"\n    Sets the seed for reproducibility in NumPy, random, and PyTorch.\n    \"\"\"\n    np.random.seed(seed)\n    random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(seed)\n        torch.backends.cudnn.deterministic = True\n        torch.backends.cudnn.benchmark = False\n\n# Set seed for reproducibility\nset_seed(42)\n\nprint(f\"Using device: {'GPU' if torch.cuda.is_available() else 'CPU'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:43.228648Z","iopub.execute_input":"2025-01-22T11:56:43.228852Z","iopub.status.idle":"2025-01-22T11:56:52.714202Z","shell.execute_reply.started":"2025-01-22T11:56:43.228832Z","shell.execute_reply":"2025-01-22T11:56:52.713186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"# GENERAL CONFIG\neda = False\ndebug = False\nnum_labels = 2\nid_fold = 0\n\n# TOKENIZE & BATCH CONFIG\nmax_length = 160\n\nper_device_train_batch_size=128\nper_device_eval_batch_size=128\n\n\n# MODEL CONFIG\nmodel_name = \"bert-base-uncased\"\nlearning_rate=2e-5\nnum_train_epochs=3 if debug == False else 10\nweight_decay=0.01\ngradient_accumulation_steps=4\nfp16=True\nmetric_for_best_model=\"f1\"\n\n# OTHERS\noutput_dir= f\"./results/fold{id_fold}\"  # Directory to save checkpoints and results\nlogging_dir=\"./logs\"  # Directory to save logs\neval_strategy=\"epoch\"\nsave_strategy=\"epoch\"\nlogging_steps= 1000 if debug == False else 10\nload_best_model_at_end=True\nreport_to=\"none\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:52.715163Z","iopub.execute_input":"2025-01-22T11:56:52.716049Z","iopub.status.idle":"2025-01-22T11:56:52.721915Z","shell.execute_reply.started":"2025-01-22T11:56:52.715966Z","shell.execute_reply":"2025-01-22T11:56:52.720880Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\nif debug:\n    train = train[:2000]\n    test_df = test_df[:1000]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:52.722744Z","iopub.execute_input":"2025-01-22T11:56:52.723068Z","iopub.status.idle":"2025-01-22T11:56:56.704222Z","shell.execute_reply.started":"2025-01-22T11:56:52.722996Z","shell.execute_reply":"2025-01-22T11:56:56.703367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:56.705053Z","iopub.execute_input":"2025-01-22T11:56:56.705391Z","iopub.status.idle":"2025-01-22T11:56:56.718373Z","shell.execute_reply.started":"2025-01-22T11:56:56.705366Z","shell.execute_reply":"2025-01-22T11:56:56.717423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"There are:\",len(train),\"samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:56.720815Z","iopub.execute_input":"2025-01-22T11:56:56.721121Z","iopub.status.idle":"2025-01-22T11:56:56.729147Z","shell.execute_reply.started":"2025-01-22T11:56:56.721098Z","shell.execute_reply":"2025-01-22T11:56:56.728338Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:56.730819Z","iopub.execute_input":"2025-01-22T11:56:56.731142Z","iopub.status.idle":"2025-01-22T11:56:56.750596Z","shell.execute_reply.started":"2025-01-22T11:56:56.731117Z","shell.execute_reply":"2025-01-22T11:56:56.749641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"No duplicated in id","metadata":{}},{"cell_type":"code","source":"train[\"num_word\"] = train[\"question_text\"].apply(lambda x: len(str(x).split()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:56.751787Z","iopub.execute_input":"2025-01-22T11:56:56.752160Z","iopub.status.idle":"2025-01-22T11:56:56.764795Z","shell.execute_reply.started":"2025-01-22T11:56:56.752137Z","shell.execute_reply":"2025-01-22T11:56:56.763757Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configure plot\nplt.figure(figsize=(16, 6))\nax = plt.gca()\n\n# Create histogram bins for each possible word length\ncounts, bins, patches = plt.hist(train[\"num_word\"], \n                                bins=np.arange(0, 140) - 0.5,  # Center bars on integer values\n                                edgecolor='black', \n                                alpha=0.7,\n                                color='#1f77b4')\n\n# Formatting\nplt.title('Distribution of Word Counts (0-160 words)', fontsize=14, pad=20)\nplt.xlabel('Number of Words', fontsize=12)\nplt.ylabel('Number of Texts', fontsize=12)\nplt.xlim(0, 140)\n\n# Add labels for every 5 words\nbin_centers = 0.5 * (bins[:-1] + bins[1:])\nfor i, (count, bin_center) in enumerate(zip(counts, bin_centers)):\n    if bin_center % 5 == 0:  # Label every 5 units\n        ax.text(bin_center, count + 500, f'{int(bin_center)}', \n                ha='center', va='bottom', rotation=90, fontsize=8)\n\n# Add statistics box\nstats_text = f\"\"\"Total Samples: {len(train):,}\nMean: {train['num_word'].mean():.1f}\nMedian: {train['num_word'].median()}\nMax: {train['num_word'].max()}\n95th %ile: {np.percentile(train['num_word'], 95)}\"\"\"\nplt.gcf().text(0.92, 0.6, stats_text, bbox=dict(facecolor='white', alpha=0.8), fontsize=10)\n\nplt.grid(axis='y', alpha=0.3)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:56.765776Z","iopub.execute_input":"2025-01-22T11:56:56.766155Z","iopub.status.idle":"2025-01-22T11:56:57.367094Z","shell.execute_reply.started":"2025-01-22T11:56:56.766115Z","shell.execute_reply":"2025-01-22T11:56:57.366055Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Short text (<140)","metadata":{}},{"cell_type":"markdown","source":"# Load tokenizer","metadata":{}},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(model_name)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.368141Z","iopub.execute_input":"2025-01-22T11:56:57.368583Z","iopub.status.idle":"2025-01-22T11:56:57.559002Z","shell.execute_reply.started":"2025-01-22T11:56:57.368529Z","shell.execute_reply":"2025-01-22T11:56:57.558269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Tokenizer type: {type(tokenizer).__name__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.559948Z","iopub.execute_input":"2025-01-22T11:56:57.560258Z","iopub.status.idle":"2025-01-22T11:56:57.565546Z","shell.execute_reply.started":"2025-01-22T11:56:57.560235Z","shell.execute_reply":"2025-01-22T11:56:57.564419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if eda: # take time, so only run for eda\n    train[\"token_length\"] = train[\"question_text\"].apply(\n        lambda x: len(\n            tokenizer.encode_plus(\n                x,\n                truncation=False,  # Do not truncate to ensure we capture the full length\n                add_special_tokens=True  # Include [CLS] and [SEP] tokens\n            )[\"input_ids\"]\n        )\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.566464Z","iopub.execute_input":"2025-01-22T11:56:57.566756Z","iopub.status.idle":"2025-01-22T11:56:57.582019Z","shell.execute_reply.started":"2025-01-22T11:56:57.566735Z","shell.execute_reply":"2025-01-22T11:56:57.581138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if eda:\n    print(\"longest token length:\",max(train[\"token_length\"]))\n    display(train[train[\"token_length\"]>max_length])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.583053Z","iopub.execute_input":"2025-01-22T11:56:57.583410Z","iopub.status.idle":"2025-01-22T11:56:57.601473Z","shell.execute_reply.started":"2025-01-22T11:56:57.583379Z","shell.execute_reply":"2025-01-22T11:56:57.600569Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Split Kfold","metadata":{}},{"cell_type":"code","source":"# Function to create K-Folds\ndef create_folds(dataframe, n_splits=5, seed=42):\n    set_seed(seed)  # Ensure reproducibility\n    dataframe[\"fold\"] = -1  # Initialize fold column with -1\n    \n    # Shuffle the data\n    dataframe = dataframe.sample(frac=1, random_state=seed).reset_index(drop=True)\n    \n    # Initialize KFold\n    kf = KFold(n_splits=n_splits, shuffle=True, random_state=seed)\n    \n    # Assign folds\n    for fold, (_, val_idx) in enumerate(kf.split(X=dataframe)):\n        dataframe.loc[val_idx, \"fold\"] = fold\n\n    return dataframe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.602446Z","iopub.execute_input":"2025-01-22T11:56:57.602781Z","iopub.status.idle":"2025-01-22T11:56:57.618248Z","shell.execute_reply.started":"2025-01-22T11:56:57.602757Z","shell.execute_reply":"2025-01-22T11:56:57.617213Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = create_folds(train, n_splits=5, seed=42)\n\n# Select fold 0 as validation set\nval_df = train[train[\"fold\"] == id_fold]\ntrain_df = train[train[\"fold\"] != id_fold]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.619212Z","iopub.execute_input":"2025-01-22T11:56:57.619586Z","iopub.status.idle":"2025-01-22T11:56:57.645050Z","shell.execute_reply.started":"2025-01-22T11:56:57.619560Z","shell.execute_reply":"2025-01-22T11:56:57.644086Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load model and tokenizer","metadata":{}},{"cell_type":"code","source":"model_name = model_name\ntokenizer = AutoTokenizer.from_pretrained(model_name)\nmodel = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=num_labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:57.646382Z","iopub.execute_input":"2025-01-22T11:56:57.646787Z","iopub.status.idle":"2025-01-22T11:56:58.041821Z","shell.execute_reply.started":"2025-01-22T11:56:57.646752Z","shell.execute_reply":"2025-01-22T11:56:58.040821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"train_dataset = Dataset.from_pandas(train_df)\nval_dataset = Dataset.from_pandas(val_df)\n\ndef preprocess_function(examples):\n    return tokenizer(examples['question_text'], truncation=True, max_length=max_length)  # No padding here\n\ntrain_dataset = train_dataset.map(preprocess_function, batched=True)\nval_dataset = val_dataset.map(preprocess_function, batched=True)\n\ntrain_dataset = train_dataset.rename_column(\"target\", \"labels\")\nval_dataset = val_dataset.rename_column(\"target\", \"labels\")\n\ndata_collator = DataCollatorWithPadding(tokenizer=tokenizer)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:58.042979Z","iopub.execute_input":"2025-01-22T11:56:58.043356Z","iopub.status.idle":"2025-01-22T11:56:58.236178Z","shell.execute_reply.started":"2025-01-22T11:56:58.043329Z","shell.execute_reply":"2025-01-22T11:56:58.235131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dataset = Dataset.from_pandas(test_df)\ntest_dataset = test_dataset.map(preprocess_function, batched=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:59:29.162719Z","iopub.execute_input":"2025-01-22T11:59:29.163100Z","iopub.status.idle":"2025-01-22T11:59:29.265108Z","shell.execute_reply.started":"2025-01-22T11:59:29.163060Z","shell.execute_reply":"2025-01-22T11:59:29.263641Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Training","metadata":{"execution":{"iopub.status.busy":"2025-01-22T10:00:07.073996Z","iopub.execute_input":"2025-01-22T10:00:07.074324Z"}}},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=output_dir,  # Directory to save checkpoints and results\n    eval_strategy=eval_strategy,  # Evaluate at the end of each epoch\n    learning_rate=learning_rate,\n    per_device_train_batch_size=per_device_train_batch_size,\n    per_device_eval_batch_size=per_device_eval_batch_size,\n    num_train_epochs=num_train_epochs,\n    weight_decay=weight_decay,\n    logging_dir=logging_dir,  # Directory to save logs\n    logging_steps=logging_steps,\n    save_strategy=save_strategy,\n    gradient_accumulation_steps=gradient_accumulation_steps,\n    load_best_model_at_end=load_best_model_at_end,  # Load best model at the end of training\n    metric_for_best_model=metric_for_best_model,  # Use accuracy to select the best model\n    report_to=report_to,  # Disable reporting to external services\n    fp16=fp16\n)\n\ndef compute_metrics(eval_pred):\n    logits, labels = eval_pred\n    predictions = logits.argmax(axis=-1)\n    precision, recall, f1, _ = precision_recall_fscore_support(labels, predictions, average=\"binary\",zero_division=0)\n    acc = accuracy_score(labels, predictions)\n    return {\n        \"accuracy\": acc,\n        \"precision\": precision,\n        \"recall\": recall,\n        \"f1\": f1,\n    }\n\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=train_dataset,\n    eval_dataset=val_dataset,\n    processing_class=tokenizer,\n    compute_metrics=compute_metrics,\n    data_collator=data_collator\n)\n\n# Step 7: Train the Model\ntrainer.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:56:58.302378Z","iopub.execute_input":"2025-01-22T11:56:58.302754Z","iopub.status.idle":"2025-01-22T11:58:40.070255Z","shell.execute_reply.started":"2025-01-22T11:56:58.302718Z","shell.execute_reply":"2025-01-22T11:58:40.069364Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation","metadata":{}},{"cell_type":"code","source":"# Evaluate the model\nresults = trainer.evaluate()\nprint(\"Evaluation Results:\", results)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:58:40.074148Z","iopub.execute_input":"2025-01-22T11:58:40.074422Z","iopub.status.idle":"2025-01-22T11:58:40.706949Z","shell.execute_reply.started":"2025-01-22T11:58:40.074399Z","shell.execute_reply":"2025-01-22T11:58:40.705905Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"predictions = trainer.predict(test_dataset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:59:36.909119Z","iopub.execute_input":"2025-01-22T11:59:36.909453Z","iopub.status.idle":"2025-01-22T11:59:38.536578Z","shell.execute_reply.started":"2025-01-22T11:59:36.909429Z","shell.execute_reply":"2025-01-22T11:59:38.535651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predicted_labels = predictions.predictions.argmax(axis=-1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:59:40.181281Z","iopub.execute_input":"2025-01-22T11:59:40.181664Z","iopub.status.idle":"2025-01-22T11:59:40.186409Z","shell.execute_reply.started":"2025-01-22T11:59:40.181633Z","shell.execute_reply":"2025-01-22T11:59:40.185334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extract 'qid' from the test DataFrame or Dataset\nqids = test_df[\"qid\"]  # Replace with the actual source of 'qid'\n\n# Create a DataFrame for submission\nsubmission_df = pd.DataFrame({\n    \"qid\": qids,\n    \"prediction\": predicted_labels\n})\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-22T11:59:41.270387Z","iopub.execute_input":"2025-01-22T11:59:41.270759Z","iopub.status.idle":"2025-01-22T11:59:41.282298Z","shell.execute_reply.started":"2025-01-22T11:59:41.270726Z","shell.execute_reply":"2025-01-22T11:59:41.281364Z"}},"outputs":[],"execution_count":null}]}