{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"},{"sourceId":2580,"sourceType":"modelInstanceVersion","modelInstanceId":1882,"modelId":244},{"sourceId":2938,"sourceType":"modelInstanceVersion","modelInstanceId":2180,"modelId":244}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport tensorflow_hub as hub\nimport tensorflow_text as text\nimport tf_keras as keras\nimport tensorflow as tf\n\nfrom tf_keras.callbacks import EarlyStopping, ReduceLROnPlateau, ModelCheckpoint\nfor item in os.walk(\"/kaggle/input/quora-insincere-questions-classification\"):\n    print(item)\n\npath_list = list(list(os.walk(\"/kaggle/input/quora-insincere-questions-classification\"))[0])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:00.256186Z","iopub.execute_input":"2025-05-13T09:32:00.256456Z","iopub.status.idle":"2025-05-13T09:32:16.558011Z","shell.execute_reply.started":"2025-05-13T09:32:00.256431Z","shell.execute_reply":"2025-05-13T09:32:16.557210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"del path_list[1]\n\nprint(path_list)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:16.559241Z","iopub.execute_input":"2025-05-13T09:32:16.559704Z","iopub.status.idle":"2025-05-13T09:32:16.563686Z","shell.execute_reply.started":"2025-05-13T09:32:16.559684Z","shell.execute_reply":"2025-05-13T09:32:16.562889Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission = pd.read_csv(os.path.join(path_list[0], path_list[1][0]))\nlen(sample_submission), sample_submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:16.564514Z","iopub.execute_input":"2025-05-13T09:32:16.564784Z","iopub.status.idle":"2025-05-13T09:32:16.888967Z","shell.execute_reply.started":"2025-05-13T09:32:16.564760Z","shell.execute_reply":"2025-05-13T09:32:16.888177Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(path_list[0], path_list[1][2])).drop(\"qid\", axis=1)\nlen(train_df), train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:16.889783Z","iopub.execute_input":"2025-05-13T09:32:16.889989Z","iopub.status.idle":"2025-05-13T09:32:20.571869Z","shell.execute_reply.started":"2025-05-13T09:32:16.889973Z","shell.execute_reply":"2025-05-13T09:32:20.571207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:20.573619Z","iopub.execute_input":"2025-05-13T09:32:20.573844Z","iopub.status.idle":"2025-05-13T09:32:20.648817Z","shell.execute_reply.started":"2025-05-13T09:32:20.573827Z","shell.execute_reply":"2025-05-13T09:32:20.648115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"target\"].plot(kind='hist', title='Target Distribution', figsize=(4,4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:20.649697Z","iopub.execute_input":"2025-05-13T09:32:20.650310Z","iopub.status.idle":"2025-05-13T09:32:21.047244Z","shell.execute_reply.started":"2025-05-13T09:32:20.650280Z","shell.execute_reply":"2025-05-13T09:32:21.046433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.random.seed(13)\ntf.random.set_seed(13)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.047979Z","iopub.execute_input":"2025-05-13T09:32:21.048256Z","iopub.status.idle":"2025-05-13T09:32:21.052767Z","shell.execute_reply.started":"2025-05-13T09:32:21.048237Z","shell.execute_reply":"2025-05-13T09:32:21.051836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_data, val_data = train_test_split(train_df, train_size=0.008, test_size=0.002, stratify=train_df.target.values)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.053700Z","iopub.execute_input":"2025-05-13T09:32:21.054534Z","iopub.status.idle":"2025-05-13T09:32:21.677315Z","shell.execute_reply.started":"2025-05-13T09:32:21.054513Z","shell.execute_reply":"2025-05-13T09:32:21.676386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_data), train_data.head(), train_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.678416Z","iopub.execute_input":"2025-05-13T09:32:21.679116Z","iopub.status.idle":"2025-05-13T09:32:21.686025Z","shell.execute_reply.started":"2025-05-13T09:32:21.679084Z","shell.execute_reply":"2025-05-13T09:32:21.685181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(val_data), val_data.head(), val_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.686893Z","iopub.execute_input":"2025-05-13T09:32:21.687162Z","iopub.status.idle":"2025-05-13T09:32:21.701181Z","shell.execute_reply.started":"2025-05-13T09:32:21.687139Z","shell.execute_reply":"2025-05-13T09:32:21.700313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, y_train = train_data['question_text'].values, train_data['target'].values\nX_val, y_val = val_data['question_text'].values, val_data['target'].values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.701943Z","iopub.execute_input":"2025-05-13T09:32:21.702179Z","iopub.status.idle":"2025-05-13T09:32:21.712463Z","shell.execute_reply.started":"2025-05-13T09:32:21.702163Z","shell.execute_reply":"2025-05-13T09:32:21.711773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train[:5], y_train[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.713246Z","iopub.execute_input":"2025-05-13T09:32:21.713491Z","iopub.status.idle":"2025-05-13T09:32:21.726709Z","shell.execute_reply.started":"2025-05-13T09:32:21.713465Z","shell.execute_reply":"2025-05-13T09:32:21.725908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_val[:5], y_val[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.727582Z","iopub.execute_input":"2025-05-13T09:32:21.728031Z","iopub.status.idle":"2025-05-13T09:32:21.738853Z","shell.execute_reply.started":"2025-05-13T09:32:21.728006Z","shell.execute_reply":"2025-05-13T09:32:21.738116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the preprocessing layer\npreprocessor = hub.KerasLayer(\"https://kaggle.com/models/tensorflow/bert/TensorFlow2/en-uncased-preprocess/3\")\n\n# Create the encoder layer\nencoder = hub.KerasLayer(\"https://www.kaggle.com/models/tensorflow/bert/tensorFlow2/en-uncased-l-12-h-768-a-12/4\", trainable=True)\n\n# Build the model using the functional API\ntext_input = keras.layers.Input(shape=(), dtype=tf.string)\nencoder_inputs = preprocessor(text_input)\noutputs = encoder(encoder_inputs)\npool_output = outputs[\"pooled_output\"]\n\nx = keras.layers.Dropout(0.5)(pool_output)\nx = keras.layers.Dense(16, activation='relu')(x)\n\noutput = keras.layers.Dense(1, activation='sigmoid')(x)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:32:21.741731Z","iopub.execute_input":"2025-05-13T09:32:21.742013Z","iopub.status.idle":"2025-05-13T09:33:03.818944Z","shell.execute_reply.started":"2025-05-13T09:32:21.741987Z","shell.execute_reply":"2025-05-13T09:33:03.818395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = keras.Model(inputs=[text_input], outputs=[output])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:33:03.819568Z","iopub.execute_input":"2025-05-13T09:33:03.819746Z","iopub.status.idle":"2025-05-13T09:33:03.827323Z","shell.execute_reply.started":"2025-05-13T09:33:03.819732Z","shell.execute_reply":"2025-05-13T09:33:03.826636Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:33:03.828129Z","iopub.execute_input":"2025-05-13T09:33:03.828413Z","iopub.status.idle":"2025-05-13T09:33:03.870780Z","shell.execute_reply.started":"2025-05-13T09:33:03.828389Z","shell.execute_reply":"2025-05-13T09:33:03.870094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=keras.optimizers.Adam(learning_rate=2e-5), loss=keras.losses.BinaryCrossentropy(\n    label_smoothing=0.0,\n    axis=-1,\n    reduction=\"sum_over_batch_size\",\n    name=\"binary_crossentropy\"), metrics=[\"accuracy\"])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:33:03.871565Z","iopub.execute_input":"2025-05-13T09:33:03.871805Z","iopub.status.idle":"2025-05-13T09:33:03.889542Z","shell.execute_reply.started":"2025-05-13T09:33:03.871779Z","shell.execute_reply":"2025-05-13T09:33:03.889031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stopping = EarlyStopping(monitor=\"val_loss\", patience=3, verbose=1)\nreduce_lr = ReduceLROnPlateau(monitor=\"val_loss\", factor=0.5, patience=2, min_lr=1e-6, verbose=1)\ncheckpoint = ModelCheckpoint(filepath=\"/kaggle/working/Best-Quora-Questions-Classifier.weights.h5\", monitor='val_loss', mode='min', save_best_only=True, save_weights_only=True, verbose=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:33:03.890516Z","iopub.execute_input":"2025-05-13T09:33:03.890691Z","iopub.status.idle":"2025-05-13T09:33:03.894685Z","shell.execute_reply.started":"2025-05-13T09:33:03.890677Z","shell.execute_reply":"2025-05-13T09:33:03.893893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n    X_train, y_train,\n    validation_data = (X_val, y_val),\n    epochs=5,\n    verbose=1,\n    callbacks=[early_stopping, reduce_lr, checkpoint]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T09:33:03.895334Z","iopub.execute_input":"2025-05-13T09:33:03.895578Z","iopub.status.idle":"2025-05-13T09:56:59.401472Z","shell.execute_reply.started":"2025-05-13T09:33:03.895561Z","shell.execute_reply":"2025-05-13T09:56:59.400682Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_weights(\"/kaggle/working/Best-Quora-Questions-Classifier.weights.h5\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:02:52.328779Z","iopub.execute_input":"2025-05-13T10:02:52.329322Z","iopub.status.idle":"2025-05-13T10:02:53.701409Z","shell.execute_reply.started":"2025-05-13T10:02:52.329297Z","shell.execute_reply":"2025-05-13T10:02:53.700733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Plot training & validation loss\nplt.figure(figsize=(12, 5))\n\n# Loss Graph\nplt.subplot(1, 2, 1)\nplt.plot(history.history['loss'], label='Loss')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title(\"Training Loss\")\nplt.legend()\n\n# Accuracy Graph\nplt.subplot(1, 2, 2)\nplt.plot(history.history['accuracy'], label='Accuracy')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"Training Accuracy\")\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:03:36.526971Z","iopub.execute_input":"2025-05-13T10:03:36.527667Z","iopub.status.idle":"2025-05-13T10:03:36.824479Z","shell.execute_reply.started":"2025-05-13T10:03:36.527647Z","shell.execute_reply":"2025-05-13T10:03:36.823720Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot training & validation loss\nplt.figure(figsize=(12, 5))\n\n# Loss Graph\nplt.subplot(1, 2, 1)\nplt.plot(history.history['val_loss'], label='Loss')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Loss\")\nplt.title(\"Validation Loss\")\nplt.legend()\n\n# Accuracy Graph\nplt.subplot(1, 2, 2)\nplt.plot(history.history['val_accuracy'], label='Accuracy')\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Accuracy\")\nplt.title(\"Validation Accuracy\")\nplt.legend()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:03:58.768526Z","iopub.execute_input":"2025-05-13T10:03:58.769255Z","iopub.status.idle":"2025-05-13T10:03:59.071166Z","shell.execute_reply.started":"2025-05-13T10:03:58.769228Z","shell.execute_reply":"2025-05-13T10:03:59.070227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_val, y_val, batch_size=32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:04:23.783031Z","iopub.execute_input":"2025-05-13T10:04:23.783311Z","iopub.status.idle":"2025-05-13T10:04:56.398803Z","shell.execute_reply.started":"2025-05-13T10:04:23.783287Z","shell.execute_reply":"2025-05-13T10:04:56.398166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\nbest_f1 = best_threshold = 0\n\ny_pred = model.predict(X_val)\n\nfor t in np.arange(0.3, 0.8, 0.05):\n    y_pred_discrete = [1 if y > t else 0 for y in y_pred]\n    f1 = f1_score(y_val, y_pred_discrete)\n    \n    if f1 > best_f1:\n        best_f1 = f1\n        best_threshold = t\n\nprint(f\"Best Threshold: {best_threshold:.2f}, Best F1 Score: {best_f1:.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:09:11.164408Z","iopub.execute_input":"2025-05-13T10:09:11.164701Z","iopub.status.idle":"2025-05-13T10:09:42.067658Z","shell.execute_reply.started":"2025-05-13T10:09:11.164679Z","shell.execute_reply":"2025-05-13T10:09:42.066928Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import classification_report\n\ny_pred = [1 if y > best_threshold else 0 for y in y_pred]\n\nprint(classification_report(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:10:01.696246Z","iopub.execute_input":"2025-05-13T10:10:01.696808Z","iopub.status.idle":"2025-05-13T10:10:01.713538Z","shell.execute_reply.started":"2025-05-13T10:10:01.696785Z","shell.execute_reply":"2025-05-13T10:10:01.712958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\nlen(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:14:51.323412Z","iopub.execute_input":"2025-05-13T10:14:51.323917Z","iopub.status.idle":"2025-05-13T10:14:51.974477Z","shell.execute_reply.started":"2025-05-13T10:14:51.323895Z","shell.execute_reply":"2025-05-13T10:14:51.973566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"questions = X_test[\"question_text\"].values\nlen(questions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:14:58.850345Z","iopub.execute_input":"2025-05-13T10:14:58.850933Z","iopub.status.idle":"2025-05-13T10:14:58.855787Z","shell.execute_reply.started":"2025-05-13T10:14:58.850910Z","shell.execute_reply":"2025-05-13T10:14:58.855205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = model.predict(questions[:1000])\n\npredictions = [1 if y > best_threshold else 0 for y in predictions]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:15:23.231966Z","iopub.execute_input":"2025-05-13T10:15:23.232274Z","iopub.status.idle":"2025-05-13T10:15:34.669774Z","shell.execute_reply.started":"2025-05-13T10:15:23.232252Z","shell.execute_reply":"2025-05-13T10:15:34.669065Z"}},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"df_submit = pd.DataFrame({\n    'qid': X_test[\"qid\"].values,\n    'prediction': predictions\n})\n\ndf_submit.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{}},{"cell_type":"code","source":"df_submit = pd.DataFrame({\n    'qid': X_test[\"qid\"][:1000].values,\n    'prediction': predictions\n})\n\ndf_submit.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-13T10:16:19.917008Z","iopub.execute_input":"2025-05-13T10:16:19.917248Z","iopub.status.idle":"2025-05-13T10:16:19.923784Z","shell.execute_reply.started":"2025-05-13T10:16:19.917232Z","shell.execute_reply":"2025-05-13T10:16:19.923224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}