{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"},{"sourceId":1246668,"sourceType":"datasetVersion","datasetId":715814}],"dockerImageVersionId":30733,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Embedding, LSTM, Dense\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom sklearn.metrics import accuracy_score\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-27T20:00:25.60428Z","iopub.execute_input":"2024-06-27T20:00:25.604907Z","iopub.status.idle":"2024-06-27T20:00:39.00571Z","shell.execute_reply.started":"2024-06-27T20:00:25.604879Z","shell.execute_reply":"2024-06-27T20:00:39.004816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\ndf_train","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:00:39.007395Z","iopub.execute_input":"2024-06-27T20:00:39.008017Z","iopub.status.idle":"2024-06-27T20:00:41.503102Z","shell.execute_reply.started":"2024-06-27T20:00:39.00799Z","shell.execute_reply":"2024-06-27T20:00:41.502193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_valid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ndf_valid","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:00:41.50424Z","iopub.execute_input":"2024-06-27T20:00:41.504769Z","iopub.status.idle":"2024-06-27T20:00:41.618199Z","shell.execute_reply.started":"2024-06-27T20:00:41.504726Z","shell.execute_reply":"2024-06-27T20:00:41.617134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_features = 20000 \nmaxlen = 100 \nembedding_size = 100 ","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:00:41.620984Z","iopub.execute_input":"2024-06-27T20:00:41.621308Z","iopub.status.idle":"2024-06-27T20:00:41.625858Z","shell.execute_reply.started":"2024-06-27T20:00:41.62128Z","shell.execute_reply":"2024-06-27T20:00:41.62479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(pd.concat([df_train['comment_text']]))","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:00:41.627299Z","iopub.execute_input":"2024-06-27T20:00:41.627688Z","iopub.status.idle":"2024-06-27T20:00:57.921653Z","shell.execute_reply.started":"2024-06-27T20:00:41.627659Z","shell.execute_reply":"2024-06-27T20:00:57.920828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_seq = tokenizer.texts_to_sequences(df_train['comment_text'])","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:00:57.922644Z","iopub.execute_input":"2024-06-27T20:00:57.922913Z","iopub.status.idle":"2024-06-27T20:01:10.782964Z","shell.execute_reply.started":"2024-06-27T20:00:57.922891Z","shell.execute_reply":"2024-06-27T20:01:10.781873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pad_sequences(train_seq, maxlen=maxlen)","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:01:10.784229Z","iopub.execute_input":"2024-06-27T20:01:10.784603Z","iopub.status.idle":"2024-06-27T20:01:12.252247Z","shell.execute_reply.started":"2024-06-27T20:01:10.784572Z","shell.execute_reply":"2024-06-27T20:01:12.251492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = df_train['toxic'].values","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:01:12.253298Z","iopub.execute_input":"2024-06-27T20:01:12.253586Z","iopub.status.idle":"2024-06-27T20:01:12.258145Z","shell.execute_reply.started":"2024-06-27T20:01:12.253562Z","shell.execute_reply":"2024-06-27T20:01:12.257155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the GloVe embeddings\nembeddings_index = {}\nwith open(\"/kaggle/input/glove6b100dtxt/glove.6B.100d.txt\", 'r', encoding='utf8') as f:\n    for line in f:\n        values = line.split()\n        word = values[0]\n        coefs = np.asarray(values[1:], dtype='float32')\n        embeddings_index[word] = coefs\n\n# Preparing the embedding matrix\nembedding_matrix = np.zeros((max_features, embedding_size))\nfor word, i in tokenizer.word_index.items():\n    if i < max_features:\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:01:12.259197Z","iopub.execute_input":"2024-06-27T20:01:12.259499Z","iopub.status.idle":"2024-06-27T20:01:24.606256Z","shell.execute_reply.started":"2024-06-27T20:01:12.259467Z","shell.execute_reply":"2024-06-27T20:01:24.605415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for test_size in [0.4, 0.3, 0.2, 0.1]:\n    validation_seq = tokenizer.texts_to_sequences(df_valid['comment_text'].iloc[:int(df_valid.shape[0] * test_size)])\n    validation_data = pad_sequences(validation_seq, maxlen=maxlen)\n    validation_labels = df_valid['toxic'].iloc[:int(df_valid.shape[0] * test_size)].values\n    model = Sequential()\n    model.add(Embedding(max_features, embedding_size, weights=[embedding_matrix], input_length=maxlen, trainable=False))\n    model.add(LSTM(64, dropout=0.2, recurrent_dropout=0.2, return_sequences=True))\n    model.add(LSTM(32, dropout=0.2, recurrent_dropout=0.2))\n    model.add(Dense(256, activation='elu'))\n    model.add(Dense(1, activation='sigmoid'))\n    model.compile(optimizer='adam', loss='binary_crossentropy')\n    model.fit(train_data, train_labels, batch_size=4096, epochs=10, verbose=0)\n    prediction = np.where(model.predict(validation_data) >= 0.5, 1, 0)\n    print(f'test size: {test_size} accuracy score: {accuracy_score(validation_labels, prediction)}')","metadata":{"execution":{"iopub.status.busy":"2024-06-27T20:05:30.187276Z","iopub.execute_input":"2024-06-27T20:05:30.187907Z","iopub.status.idle":"2024-06-27T20:08:00.529899Z","shell.execute_reply.started":"2024-06-27T20:05:30.187875Z","shell.execute_reply":"2024-06-27T20:08:00.528393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}