{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":2580,"sourceType":"modelInstanceVersion","modelInstanceId":1882,"modelId":244},{"sourceId":2938,"sourceType":"modelInstanceVersion","modelInstanceId":2180,"modelId":244}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Obtaining Testing and Training data","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.metrics import classification_report, confusion_matrix, f1_score, accuracy_score, roc_curve, auc\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport xgboost as xgb\n\nnp.random.seed(42)\ndf = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ndf = df.drop(\"qid\", axis=1)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:06:40.196949Z","iopub.execute_input":"2025-04-16T18:06:40.197266Z","iopub.status.idle":"2025-04-16T18:06:42.767954Z","shell.execute_reply.started":"2025-04-16T18:06:40.197240Z","shell.execute_reply":"2025-04-16T18:06:42.767006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:06:45.367615Z","iopub.execute_input":"2025-04-16T18:06:45.368037Z","iopub.status.idle":"2025-04-16T18:06:45.391126Z","shell.execute_reply.started":"2025-04-16T18:06:45.368001Z","shell.execute_reply":"2025-04-16T18:06:45.390261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Vectorize Sentences","metadata":{}},{"cell_type":"code","source":"from imblearn.under_sampling import RandomUnderSampler\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nX = df[\"question_text\"]\ny = df[\"target\"]\n\nvectorizer = TfidfVectorizer(\n    max_features=10000,\n    ngram_range=(1, 2),\n    stop_words=\"english\"\n)\n\nX_tfidf = vectorizer.fit_transform(X)\n\nrus = RandomUnderSampler(sampling_strategy='auto', random_state=42)\nX_resampled, y_resampled = rus.fit_resample(X_tfidf, y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:06:53.631622Z","iopub.execute_input":"2025-04-16T18:06:53.631946Z","iopub.status.idle":"2025-04-16T18:07:33.299848Z","shell.execute_reply.started":"2025-04-16T18:06:53.631891Z","shell.execute_reply":"2025-04-16T18:07:33.299151Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train and Validation split","metadata":{}},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X_resampled, y_resampled, random_state=42, stratify=y_resampled)\n\nX_test = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\nx_test = X_test[\"question_text\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:07:37.796027Z","iopub.execute_input":"2025-04-16T18:07:37.796480Z","iopub.status.idle":"2025-04-16T18:07:39.040706Z","shell.execute_reply.started":"2025-04-16T18:07:37.796452Z","shell.execute_reply":"2025-04-16T18:07:39.039704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.utils.class_weight import compute_class_weight\nimport numpy as np\n\nclassWeights = compute_class_weight(\n    class_weight='balanced',\n    classes=np.unique(y_train),\n    y=y_train\n)\n\nclassWeightDict = dict(enumerate(classWeights))\nprint(classWeightDict)\nplt.figure(figsize=(6, 4))\nplt.bar(classWeightDict.keys(), classWeightDict.values(), color='skyblue')\nplt.xlabel('Class')\nplt.ylabel('Weight')\nplt.title('Class Weights')\nplt.xticks(list(classWeightDict.keys()))\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:07:42.561212Z","iopub.execute_input":"2025-04-16T18:07:42.561515Z","iopub.status.idle":"2025-04-16T18:07:42.823633Z","shell.execute_reply.started":"2025-04-16T18:07:42.561489Z","shell.execute_reply":"2025-04-16T18:07:42.822797Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Model Definition","metadata":{}},{"cell_type":"code","source":"model = xgb.XGBClassifier(\n    objective=\"binary:logistic\",\n    eval_metric=\"logloss\",\n    use_label_encoder=False,\n    n_jobs=-1,\n    random_state=42,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:07:47.762747Z","iopub.execute_input":"2025-04-16T18:07:47.763093Z","iopub.status.idle":"2025-04-16T18:07:47.767213Z","shell.execute_reply.started":"2025-04-16T18:07:47.763063Z","shell.execute_reply":"2025-04-16T18:07:47.766056Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"model.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:07:50.922983Z","iopub.execute_input":"2025-04-16T18:07:50.923302Z","iopub.status.idle":"2025-04-16T18:08:13.268357Z","shell.execute_reply.started":"2025-04-16T18:07:50.923274Z","shell.execute_reply":"2025-04-16T18:08:13.267402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, confusion_matrix, roc_auc_score, roc_curve\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\ny_pred = model.predict(X_val)\ny_pred_proba = model.predict_proba(X_val)[:, 1]\n\naccuracy = accuracy_score(y_val, y_pred)\nprint(f\"Validation Accuracy: {accuracy:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:08:19.893053Z","iopub.execute_input":"2025-04-16T18:08:19.893337Z","iopub.status.idle":"2025-04-16T18:08:20.035651Z","shell.execute_reply.started":"2025-04-16T18:08:19.893316Z","shell.execute_reply":"2025-04-16T18:08:20.034010Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_val, y_pred)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted Label\")\nplt.ylabel(\"True Label\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:08:23.420335Z","iopub.execute_input":"2025-04-16T18:08:23.420635Z","iopub.status.idle":"2025-04-16T18:08:23.600779Z","shell.execute_reply.started":"2025-04-16T18:08:23.420612Z","shell.execute_reply":"2025-04-16T18:08:23.599748Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"cell_type":"code","source":"roc_auc = roc_auc_score(y_val, y_pred_proba)\nfpr, tpr, thresholds = roc_curve(y_val, y_pred_proba)\n\nprint(f\"ROC AUC Score: {roc_auc:.4f}\")\n\nplt.figure(figsize=(6, 5))\nplt.plot(fpr, tpr, label=f\"AUC = {roc_auc:.4f}\")\nplt.plot([0, 1], [0, 1], linestyle='--', color='gray')\nplt.title(\"ROC Curve\")\nplt.xlabel(\"False Positive Rate\")\nplt.ylabel(\"True Positive Rate\")\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:08:31.808397Z","iopub.execute_input":"2025-04-16T18:08:31.808700Z","iopub.status.idle":"2025-04-16T18:08:32.027784Z","shell.execute_reply.started":"2025-04-16T18:08:31.808677Z","shell.execute_reply":"2025-04-16T18:08:32.026656Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"test_tfidf = vectorizer.fit_transform(x_test)\n\ndef submit():\n    submission = X_test[['qid']].copy()\n    prediction = model.predict_proba(test_tfidf)[:, 1]\n    pred = (prediction > 0.5).astype(int)\n    submission['prediction'] = pred\n    submission.to_csv('submission.csv', index=False)\n    return submission\n\nsubmit()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T18:08:37.149556Z","iopub.execute_input":"2025-04-16T18:08:37.149925Z","iopub.status.idle":"2025-04-16T18:08:49.791500Z","shell.execute_reply.started":"2025-04-16T18:08:37.149880Z","shell.execute_reply":"2025-04-16T18:08:49.790038Z"}},"outputs":[],"execution_count":null}]}