{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-11T12:37:03.596551Z","iopub.execute_input":"2024-02-11T12:37:03.597439Z","iopub.status.idle":"2024-02-11T12:37:04.520063Z","shell.execute_reply.started":"2024-02-11T12:37:03.597404Z","shell.execute_reply":"2024-02-11T12:37:04.518882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -U sentence-transformers -q\n!pip install -U transformers","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:37:04.521903Z","iopub.execute_input":"2024-02-11T12:37:04.522446Z","iopub.status.idle":"2024-02-11T12:37:40.128593Z","shell.execute_reply.started":"2024-02-11T12:37:04.522407Z","shell.execute_reply":"2024-02-11T12:37:40.127459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm\nfrom transformers import AutoTokenizer, DataCollatorWithPadding, AutoModelForSequenceClassification, TrainingArguments, Trainer\nfrom datasets import Dataset\nfrom sklearn.model_selection import train_test_split\n# from fastai.text.all import *\nimport torch\nfrom sentence_transformers import SentenceTransformer\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:37:40.129957Z","iopub.execute_input":"2024-02-11T12:37:40.130306Z","iopub.status.idle":"2024-02-11T12:37:58.464929Z","shell.execute_reply.started":"2024-02-11T12:37:40.130274Z","shell.execute_reply":"2024-02-11T12:37:58.463978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:37:58.467483Z","iopub.execute_input":"2024-02-11T12:37:58.468564Z","iopub.status.idle":"2024-02-11T12:38:03.779487Z","shell.execute_reply.started":"2024-02-11T12:37:58.468526Z","shell.execute_reply":"2024-02-11T12:38:03.778347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:38:03.780809Z","iopub.execute_input":"2024-02-11T12:38:03.781275Z","iopub.status.idle":"2024-02-11T12:38:03.800880Z","shell.execute_reply.started":"2024-02-11T12:38:03.781242Z","shell.execute_reply":"2024-02-11T12:38:03.799170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_set = list(train_df['question_text'].head(10000))\nall_labels=list(train_df['target'].head(10000))\n","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:38:03.802288Z","iopub.execute_input":"2024-02-11T12:38:03.802739Z","iopub.status.idle":"2024-02-11T12:38:03.814366Z","shell.execute_reply.started":"2024-02-11T12:38:03.802705Z","shell.execute_reply":"2024-02-11T12:38:03.812629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_st1 = SentenceTransformer('all-mpnet-base-v2')\nmodel_st2 = SentenceTransformer('all-MiniLM-L6-v2')\nmodel_st3 = SentenceTransformer('paraphrase-mpnet-base-v2')","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:38:03.815834Z","iopub.execute_input":"2024-02-11T12:38:03.816705Z","iopub.status.idle":"2024-02-11T12:39:02.505635Z","shell.execute_reply.started":"2024-02-11T12:38:03.816670Z","shell.execute_reply":"2024-02-11T12:39:02.504615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def embed(model,sentences):\n    embeddings=model.encode(sentences)\n    return embeddings","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:39:02.506886Z","iopub.execute_input":"2024-02-11T12:39:02.507201Z","iopub.status.idle":"2024-02-11T12:39:02.511795Z","shell.execute_reply.started":"2024-02-11T12:39:02.507175Z","shell.execute_reply":"2024-02-11T12:39:02.510851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_st1 = embed(model_st1, all_set)\nembeddings_st2 = embed(model_st2, all_set)\nembeddings_st3 = embed(model_st3, all_set)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:39:02.512924Z","iopub.execute_input":"2024-02-11T12:39:02.513291Z","iopub.status.idle":"2024-02-11T12:39:24.652883Z","shell.execute_reply.started":"2024-02-11T12:39:02.513267Z","shell.execute_reply":"2024-02-11T12:39:24.651843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.manifold import TSNE\nX_embedded = TSNE(n_components=2).fit_transform(embeddings_st1)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:39:24.656681Z","iopub.execute_input":"2024-02-11T12:39:24.656987Z","iopub.status.idle":"2024-02-11T12:40:21.410640Z","shell.execute_reply.started":"2024-02-11T12:39:24.656961Z","shell.execute_reply":"2024-02-11T12:40:21.409624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_embedded.shape","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:40:21.411861Z","iopub.execute_input":"2024-02-11T12:40:21.412177Z","iopub.status.idle":"2024-02-11T12:40:21.419158Z","shell.execute_reply.started":"2024-02-11T12:40:21.412151Z","shell.execute_reply":"2024-02-11T12:40:21.418297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp=train_df.head(10000)\ntemp[\"X_comp\"]=X_embedded[:,0]\ntemp[\"Y_comp\"]=X_embedded[:,1]","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:40:21.420454Z","iopub.execute_input":"2024-02-11T12:40:21.421165Z","iopub.status.idle":"2024-02-11T12:40:21.430519Z","shell.execute_reply.started":"2024-02-11T12:40:21.421141Z","shell.execute_reply":"2024-02-11T12:40:21.429605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n    \nfig = px.scatter(temp, x=\"X_comp\", y=\"Y_comp\",  color = \"target\", size_max=60)\nfig.update_layout(\n     height=800)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:40:21.431390Z","iopub.execute_input":"2024-02-11T12:40:21.431637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.cluster import DBSCAN\nclusters = DBSCAN(eps=30,min_samples=6).fit(temp[[\"X_comp\",\"Y_comp\"]])\n# get cluster labels\nclusters.labels_","metadata":{"execution":{"iopub.status.idle":"2024-02-11T12:40:23.853895Z","shell.execute_reply.started":"2024-02-11T12:40:23.391873Z","shell.execute_reply":"2024-02-11T12:40:23.852829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from collections import Counter\nCounter(clusters.labels_)","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:40:23.855142Z","iopub.execute_input":"2024-02-11T12:40:23.855524Z","iopub.status.idle":"2024-02-11T12:40:23.866176Z","shell.execute_reply.started":"2024-02-11T12:40:23.855495Z","shell.execute_reply":"2024-02-11T12:40:23.865115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\np = sns.scatterplot(data = temp, x = \"X_comp\", y = \"Y_comp\", hue = clusters.labels_, legend = \"full\", palette = \"deep\")\nsns.move_legend(p, \"upper right\", bbox_to_anchor = (1.17, 1.), title = 'Clusters')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-11T12:40:23.867459Z","iopub.execute_input":"2024-02-11T12:40:23.867787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def embeddings(embedding):\n    X_embedded = TSNE(n_components=2).fit_transform(embedding)\n    temp=train_df.head(10000)\n    temp[\"X_comp\"]=X_embedded[:,0]\n    temp[\"Y_comp\"]=X_embedded[:,1]\n    clusters = DBSCAN(eps=30,min_samples=6).fit(temp[[\"X_comp\",\"Y_comp\"]])\n    p = sns.scatterplot(data = temp, x = \"X_comp\", y = \"Y_comp\", hue = clusters.labels_, legend = \"full\", palette = \"deep\")\n    sns.move_legend(p, \"upper right\", bbox_to_anchor = (1.17, 1.), title = 'Clusters')\n    plt.show()\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for emd in [embeddings_st1,embeddings_st2,embeddings_st3]:\n    embeddings(embeddings_st1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}