{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-03T21:25:53.703270Z","iopub.execute_input":"2022-07-03T21:25:53.703668Z","iopub.status.idle":"2022-07-03T21:25:53.714988Z","shell.execute_reply.started":"2022-07-03T21:25:53.703632Z","shell.execute_reply":"2022-07-03T21:25:53.714060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install nlpaug","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:25:53.736547Z","iopub.execute_input":"2022-07-03T21:25:53.736872Z","iopub.status.idle":"2022-07-03T21:26:01.641004Z","shell.execute_reply.started":"2022-07-03T21:25:53.736841Z","shell.execute_reply":"2022-07-03T21:26:01.640107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:01.642985Z","iopub.execute_input":"2022-07-03T21:26:01.643256Z","iopub.status.idle":"2022-07-03T21:26:01.653296Z","shell.execute_reply.started":"2022-07-03T21:26:01.643223Z","shell.execute_reply":"2022-07-03T21:26:01.652240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"global_seed = 42\n","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:01.654798Z","iopub.execute_input":"2022-07-03T21:26:01.655437Z","iopub.status.idle":"2022-07-03T21:26:01.663490Z","shell.execute_reply.started":"2022-07-03T21:26:01.655387Z","shell.execute_reply":"2022-07-03T21:26:01.662692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\ntest_df = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:01.666183Z","iopub.execute_input":"2022-07-03T21:26:01.666434Z","iopub.status.idle":"2022-07-03T21:26:03.982405Z","shell.execute_reply.started":"2022-07-03T21:26:01.666406Z","shell.execute_reply":"2022-07-03T21:26:03.981789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'] = (train_df[['toxic','severe_toxic','obscene','threat','insult','identity_hate']].sum(axis=1) >= 1).astype(int)\ntrain_df[train_df['target']==1].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:03.983582Z","iopub.execute_input":"2022-07-03T21:26:03.983923Z","iopub.status.idle":"2022-07-03T21:26:04.022236Z","shell.execute_reply.started":"2022-07-03T21:26:03.983893Z","shell.execute_reply":"2022-07-03T21:26:04.021445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(columns = ['toxic','severe_toxic','obscene','threat','insult','identity_hate'],inplace = True)\ntrain_df.set_index('id', drop = True, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:04.023475Z","iopub.execute_input":"2022-07-03T21:26:04.023813Z","iopub.status.idle":"2022-07-03T21:26:04.046327Z","shell.execute_reply.started":"2022-07-03T21:26:04.023784Z","shell.execute_reply":"2022-07-03T21:26:04.045262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df['target']==1].head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:04.048023Z","iopub.execute_input":"2022-07-03T21:26:04.048497Z","iopub.status.idle":"2022-07-03T21:26:04.067080Z","shell.execute_reply.started":"2022-07-03T21:26:04.048461Z","shell.execute_reply":"2022-07-03T21:26:04.066456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Lower Casing\ntrain_df[\"comment_text\"] = train_df[\"comment_text\"].str.lower()\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:04.068446Z","iopub.execute_input":"2022-07-03T21:26:04.068862Z","iopub.status.idle":"2022-07-03T21:26:04.442804Z","shell.execute_reply.started":"2022-07-03T21:26:04.068829Z","shell.execute_reply":"2022-07-03T21:26:04.441837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Removal of Puctuations\nimport string\nPUNCT_TO_REMOVE = string.punctuation\ndef remove_punctuation(text):\n    \"\"\"custom function to remove the punctuation\"\"\"\n    text = text.replace('\\n',' ')\n    return text.translate(str.maketrans('', '', PUNCT_TO_REMOVE)) #maketrans tasnaa mapping table lil translate\n#translate tabaa lmapping table w transformi ltext\n\ntrain_df[\"comment_text\"] = train_df[\"comment_text\"].apply(lambda text: remove_punctuation(text))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:04.444450Z","iopub.execute_input":"2022-07-03T21:26:04.444873Z","iopub.status.idle":"2022-07-03T21:26:07.560183Z","shell.execute_reply.started":"2022-07-03T21:26:04.444822Z","shell.execute_reply":"2022-07-03T21:26:07.559210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Removal of Stopwords\nfrom nltk.corpus import stopwords\nSTOPWORDS = set(stopwords.words('english'))\ndef remove_stopwords(text):\n    \"\"\"custom function to remove the stopwords\"\"\"\n    return \" \".join([word for word in str(text).split() if word not in STOPWORDS])\n\ntrain_df[\"comment_text\"] = train_df[\"comment_text\"].apply(lambda text: remove_stopwords(text))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:07.563039Z","iopub.execute_input":"2022-07-03T21:26:07.563292Z","iopub.status.idle":"2022-07-03T21:26:10.835671Z","shell.execute_reply.started":"2022-07-03T21:26:07.563262Z","shell.execute_reply":"2022-07-03T21:26:10.834790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Removal of URLS\nimport re\ndef remove_urls(text):\n    url_pattern = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url_pattern.sub(r'', text)\n\ntrain_df[\"comment_text\"] = train_df[\"comment_text\"].apply(lambda text: remove_urls(text))\n\ntrain_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:10.837446Z","iopub.execute_input":"2022-07-03T21:26:10.837959Z","iopub.status.idle":"2022-07-03T21:26:12.817936Z","shell.execute_reply.started":"2022-07-03T21:26:10.837891Z","shell.execute_reply":"2022-07-03T21:26:12.817027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts(normalize=True) #normalize = True twali trajaali les pourcentages","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.819649Z","iopub.execute_input":"2022-07-03T21:26:12.821259Z","iopub.status.idle":"2022-07-03T21:26:12.832873Z","shell.execute_reply.started":"2022-07-03T21:26:12.821207Z","shell.execute_reply":"2022-07-03T21:26:12.832000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:17:56.347211Z","iopub.execute_input":"2022-07-03T22:17:56.347617Z","iopub.status.idle":"2022-07-03T22:17:56.362195Z","shell.execute_reply.started":"2022-07-03T22:17:56.347584Z","shell.execute_reply":"2022-07-03T22:17:56.361071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.pipeline import Pipeline, make_pipeline\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom sklearn.feature_extraction.text import TfidfTransformer\nfrom sklearn.naive_bayes import MultinomialNB\n\nfrom sklearn.model_selection import train_test_split\n\n#vec = CountVectorizer(stop_words='english')\n#bag_of_words = vec.fit_transform(train_df[\"comment_text\"])\n#print(vec.vocabulary_)\n#sum_words = bag_of_words.sum(axis=0)\n#words_freq = [(word, sum_words[0, idx]) for word, idx in vec.vocabulary_.items()]\n#words_freq =sorted(words_freq, key = lambda x: x[1], reverse=True)\n\n#print(words_freq[:10])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.834451Z","iopub.execute_input":"2022-07-03T21:26:12.835513Z","iopub.status.idle":"2022-07-03T21:26:12.842539Z","shell.execute_reply.started":"2022-07-03T21:26:12.835465Z","shell.execute_reply":"2022-07-03T21:26:12.841712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target'].value_counts()\nratio = train_df[train_df['target'] ==0].count()/train_df[train_df['target'] ==1].count()\nprint(ratio[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.843833Z","iopub.execute_input":"2022-07-03T21:26:12.844100Z","iopub.status.idle":"2022-07-03T21:26:12.951507Z","shell.execute_reply.started":"2022-07-03T21:26:12.844073Z","shell.execute_reply":"2022-07-03T21:26:12.950567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df=train_df[train_df.target==1].reset_index(drop=True)\ntemp_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.952778Z","iopub.execute_input":"2022-07-03T21:26:12.953014Z","iopub.status.idle":"2022-07-03T21:26:12.979321Z","shell.execute_reply.started":"2022-07-03T21:26:12.952986Z","shell.execute_reply":"2022-07-03T21:26:12.978491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text = temp_df.iloc[22467]['comment_text']\nprint(text)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.980775Z","iopub.execute_input":"2022-07-03T21:26:12.981276Z","iopub.status.idle":"2022-07-03T21:26:12.987186Z","shell.execute_reply.started":"2022-07-03T21:26:12.981231Z","shell.execute_reply":"2022-07-03T21:26:12.986527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.988442Z","iopub.execute_input":"2022-07-03T21:26:12.988844Z","iopub.status.idle":"2022-07-03T21:26:12.997257Z","shell.execute_reply.started":"2022-07-03T21:26:12.988813Z","shell.execute_reply":"2022-07-03T21:26:12.996377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nlpaug\nimport nlpaug.augmenter.word as naw\ntry : \n    train_df= pd.read_csv('train_augmented.csv')\nexcept : \n    aug = naw.SynonymAug(aug_src='wordnet',aug_max=8)\n    new_text = []\n    for i in tqdm(range(temp_df.shape[0]), desc = 'tqdm() Progress Bar'):\n        text = temp_df.iloc[i]['comment_text']\n        augmented_text = aug.augment(text, n=8)\n        new_text.append(augmented_text)\n    flatten_list = list(np.concatenate(new_text). flat)\n    new=pd.DataFrame({'comment_text': flatten_list,'target':1})\n    train_df = train_df.append(new).reset_index(drop=True)\n    train_df.to_csv('train_augmented.csv', index = False)\ntrain_df['target'].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:12.998535Z","iopub.execute_input":"2022-07-03T21:26:12.998779Z","iopub.status.idle":"2022-07-03T21:26:14.466830Z","shell.execute_reply.started":"2022-07-03T21:26:12.998751Z","shell.execute_reply":"2022-07-03T21:26:14.466027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.duplicated().sum())\ntrain_df = train_df.drop_duplicates()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:14.468310Z","iopub.execute_input":"2022-07-03T21:26:14.468762Z","iopub.status.idle":"2022-07-03T21:26:15.369294Z","shell.execute_reply.started":"2022-07-03T21:26:14.468726Z","shell.execute_reply":"2022-07-03T21:26:15.368488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df.isnull().sum())\ntrain_df.dropna(how = 'any' , inplace= True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:15.370991Z","iopub.execute_input":"2022-07-03T21:26:15.371865Z","iopub.status.idle":"2022-07-03T21:26:15.539511Z","shell.execute_reply.started":"2022-07-03T21:26:15.371816Z","shell.execute_reply":"2022-07-03T21:26:15.538101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics.pairwise import linear_kernel\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n# Creating tf-idf matrix\ntfidf = TfidfVectorizer(analyzer='word', ngram_range=(1, 1), max_df = 20 ,min_df= 2, stop_words='english')\ntfidf_matrix = tfidf.fit_transform(train_df[\"comment_text\"])\nprint(tfidf_matrix.shape)\n#cosine_similarities = linear_kernel(tfidf_matrix, tfidf_matrix)\n\n### Creating a python object of the class TfidfVectorizer\n\nX_train,X_test,y_train,y_test = train_test_split(tfidf_matrix,train_df['target'],test_size=0.2,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:15.540789Z","iopub.execute_input":"2022-07-03T21:26:15.541132Z","iopub.status.idle":"2022-07-03T21:26:34.387414Z","shell.execute_reply.started":"2022-07-03T21:26:15.541100Z","shell.execute_reply":"2022-07-03T21:26:34.386255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix,accuracy_score \nfrom sklearn.metrics import classification_report\n\nfrom sklearn.naive_bayes import MultinomialNB\nmnb = MultinomialNB().fit(X_train, y_train) \n\ny_pred_NB=mnb.predict(X_test)\n\nprint(\"Accuracy of Multinominal Naive Balyes:\",accuracy_score(y_test, y_pred_NB))\nprint(classification_report(y_pred_NB,y_test))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:34.390037Z","iopub.execute_input":"2022-07-03T21:26:34.390400Z","iopub.status.idle":"2022-07-03T21:26:34.617945Z","shell.execute_reply.started":"2022-07-03T21:26:34.390352Z","shell.execute_reply":"2022-07-03T21:26:34.617021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction with nb","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:34.619591Z","iopub.execute_input":"2022-07-03T21:26:34.619829Z","iopub.status.idle":"2022-07-03T21:26:34.624721Z","shell.execute_reply.started":"2022-07-03T21:26:34.619802Z","shell.execute_reply":"2022-07-03T21:26:34.623512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel\nimport transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:34.626485Z","iopub.execute_input":"2022-07-03T21:26:34.627060Z","iopub.status.idle":"2022-07-03T21:26:34.635220Z","shell.execute_reply.started":"2022-07-03T21:26:34.627014Z","shell.execute_reply":"2022-07-03T21:26:34.634350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encode(texts, tokenizer, max_len=256):\n    input_ids = []\n    token_type_ids = []\n    attention_mask = []\n    \n    for text in texts:\n        token = tokenizer(text, max_length=256, truncation=True, padding='max_length',add_special_tokens=True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n        \n    return np.array(input_ids), np.array(token_type_ids), np.array(attention_mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:34.636582Z","iopub.execute_input":"2022-07-03T21:26:34.636802Z","iopub.status.idle":"2022-07-03T21:26:34.646531Z","shell.execute_reply.started":"2022-07-03T21:26:34.636776Z","shell.execute_reply":"2022-07-03T21:26:34.645635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.BertTokenizer.from_pretrained(\"bert-base-cased\")\ntokenizer.save_pretrained('.')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:34.648562Z","iopub.execute_input":"2022-07-03T21:26:34.649213Z","iopub.status.idle":"2022-07-03T21:26:35.692599Z","shell.execute_reply.started":"2022-07-03T21:26:34.649167Z","shell.execute_reply":"2022-07-03T21:26:35.691679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\ntpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:35.694117Z","iopub.execute_input":"2022-07-03T21:26:35.694452Z","iopub.status.idle":"2022-07-03T21:26:42.226479Z","shell.execute_reply.started":"2022-07-03T21:26:35.694412Z","shell.execute_reply":"2022-07-03T21:26:42.225704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(train_df['comment_text'], train_df['target'], test_size=0.12, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:42.231587Z","iopub.execute_input":"2022-07-03T21:26:42.231871Z","iopub.status.idle":"2022-07-03T21:26:42.333375Z","shell.execute_reply.started":"2022-07-03T21:26:42.231838Z","shell.execute_reply":"2022-07-03T21:26:42.332432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encode(X_train.astype(str), tokenizer)\nX_valid = bert_encode(X_valid.astype(str), tokenizer)\n\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:26:42.334955Z","iopub.execute_input":"2022-07-03T21:26:42.335179Z","iopub.status.idle":"2022-07-03T21:36:16.281553Z","shell.execute_reply.started":"2022-07-03T21:26:42.335152Z","shell.execute_reply":"2022-07-03T21:36:16.280361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\ntrain_dataset = (tf.data.Dataset.from_tensor_slices((X_train, y_train)).repeat().shuffle(2048).batch(16 * tpu_strategy.num_replicas_in_sync).prefetch(AUTO))\nvalid_dataset = (tf.data.Dataset.from_tensor_slices((X_valid, y_valid)).batch(16 * tpu_strategy.num_replicas_in_sync).cache().prefetch(AUTO))\ntrain_dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:36:16.283278Z","iopub.execute_input":"2022-07-03T21:36:16.283571Z","iopub.status.idle":"2022-07-03T21:36:23.986366Z","shell.execute_reply.started":"2022-07-03T21:36:16.283534Z","shell.execute_reply":"2022-07-03T21:36:23.985464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_model, max_len=256):    \n    input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"token_type_ids\")\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\n    sequence_output = bert_model.bert(input_ids, token_type_ids=token_type_ids, attention_mask=attention_mask)[0]\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(.1)(clf_output)\n    out = Dense(1, activation='sigmoid')(clf_output)\n    \n    model = Model(inputs=[input_ids, token_type_ids, attention_mask], outputs=out)\n    model.compile(Adam(lr = 1e-4), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:36:23.987804Z","iopub.execute_input":"2022-07-03T21:36:23.988100Z","iopub.status.idle":"2022-07-03T21:36:23.997663Z","shell.execute_reply.started":"2022-07-03T21:36:23.988069Z","shell.execute_reply":"2022-07-03T21:36:23.996699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(X_train[0]))","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:36:23.999252Z","iopub.execute_input":"2022-07-03T21:36:23.999812Z","iopub.status.idle":"2022-07-03T21:36:24.009740Z","shell.execute_reply.started":"2022-07-03T21:36:23.999764Z","shell.execute_reply":"2022-07-03T21:36:24.009040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tpu_strategy.scope():\n    transformer_layer = (TFBertModel.from_pretrained('bert-base-cased'))\n    model = build_model(transformer_layer, max_len=256)\n    \n    save_best = tf.keras.callbacks.ModelCheckpoint(\"Model.h5\", monitor='val_accuracy',save_best_only=True, verbose=2)\n    \n    model.summary()\n    print('\\n\\nModel Training..........................................\\n')\n    model.fit(train_dataset,steps_per_epoch=2618, validation_data=valid_dataset, epochs=2, callbacks=[save_best])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:36:24.011047Z","iopub.execute_input":"2022-07-03T21:36:24.011458Z","iopub.status.idle":"2022-07-03T21:50:37.088169Z","shell.execute_reply.started":"2022-07-03T21:36:24.011424Z","shell.execute_reply":"2022-07-03T21:50:37.086338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model('./Model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:50:37.095543Z","iopub.execute_input":"2022-07-03T21:50:37.095992Z","iopub.status.idle":"2022-07-03T21:50:49.930041Z","shell.execute_reply.started":"2022-07-03T21:50:37.095942Z","shell.execute_reply":"2022-07-03T21:50:49.929054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"content\"] = test_df[\"content\"].apply(lambda text: remove_punctuation(text))\ntest_df[\"content\"] = test_df[\"content\"].apply(lambda text: remove_stopwords(text))\ntest_df[\"content\"] = test_df[\"content\"].str.lower()\ntest_df[\"content\"] = test_df[\"content\"].apply(lambda text: remove_urls(text))\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:50:49.931533Z","iopub.execute_input":"2022-07-03T21:50:49.931984Z","iopub.status.idle":"2022-07-03T21:50:54.529520Z","shell.execute_reply.started":"2022-07-03T21:50:49.931911Z","shell.execute_reply":"2022-07-03T21:50:54.528592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['lang'].unique()\n#so 6 languages so maybe fil oversampling shouldve tranlated every one to the 6 other languages, bon non lzem fil target==0 zeda ntarjamhom\n#so chiwali aandi train_df.shape[0]*6 + w de même donc tawa fi blaset mantarjem l test chintarjem l train teei lkol wahda menhom\n#i think khir mil zouz ken nasmaa vecteur mil unique kelmet w minhom naamel lcategories (bon hedheka chkaaed ysir bekel tokenizer\n# so eni chnaamel ?\n#eni lezmni nvectorisi illi target == 1 weli target ==0 wahadhom w ntaramhom (ama non khayba lmethode hedhi maa lsequentiel)\n#i think meilleur methode ntarjem l train_data teei","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:11:13.342003Z","iopub.execute_input":"2022-07-03T22:11:13.342491Z","iopub.status.idle":"2022-07-03T22:11:13.355441Z","shell.execute_reply.started":"2022-07-03T22:11:13.342458Z","shell.execute_reply":"2022-07-03T22:11:13.354698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape # train data teei 400k so nope mantarjamch ltrain xD #dkhalna fil big data w jaw 400k*6 \n#so i think ntarjem kol ligne chintarjamha lil anglais seaa","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:18:48.773798Z","iopub.execute_input":"2022-07-03T22:18:48.775217Z","iopub.status.idle":"2022-07-03T22:18:48.783461Z","shell.execute_reply.started":"2022-07-03T22:18:48.775164Z","shell.execute_reply":"2022-07-03T22:18:48.782492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install translators --upgrade","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:50:54.531021Z","iopub.execute_input":"2022-07-03T21:50:54.531299Z","iopub.status.idle":"2022-07-03T21:51:03.156167Z","shell.execute_reply.started":"2022-07-03T21:50:54.531270Z","shell.execute_reply":"2022-07-03T21:51:03.154888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.dropna(how = 'any' , inplace= True)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T21:51:03.159779Z","iopub.execute_input":"2022-07-03T21:51:03.160102Z","iopub.status.idle":"2022-07-03T21:51:03.204680Z","shell.execute_reply.started":"2022-07-03T21:51:03.160065Z","shell.execute_reply":"2022-07-03T21:51:03.203841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import translators as ts\ntemp2_df = test_df.sample(20)\ntry :\n    temp2_df['comment_text'] = temp2_df['content'].apply(lambda x: ts.google(x, from_language='auto', to_language='en'))\nexcept :\n    temp3_df = test_df\n    for i in tqdm(range(temp3_df.shape[0])):\n        temp3_df['content'].iloc[i] = ts.google(temp3_df['content'].iloc[i], \n                                                from_language = temp3_df['lang'].iloc[i],\n                                                to_language='en')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:05:10.390126Z","iopub.execute_input":"2022-07-03T22:05:10.391056Z","iopub.status.idle":"2022-07-03T22:05:39.945966Z","shell.execute_reply.started":"2022-07-03T22:05:10.391004Z","shell.execute_reply":"2022-07-03T22:05:39.944860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp2_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:05:44.047150Z","iopub.execute_input":"2022-07-03T22:05:44.047833Z","iopub.status.idle":"2022-07-03T22:05:44.062875Z","shell.execute_reply.started":"2022-07-03T22:05:44.047776Z","shell.execute_reply":"2022-07-03T22:05:44.061841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_text = bert_encode(temp2_df['comment_text'].astype(str), tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:08:45.011232Z","iopub.execute_input":"2022-07-03T22:08:45.011789Z","iopub.status.idle":"2022-07-03T22:08:45.069173Z","shell.execute_reply.started":"2022-07-03T22:08:45.011754Z","shell.execute_reply":"2022-07-03T22:08:45.068282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_text[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:08:48.531159Z","iopub.execute_input":"2022-07-03T22:08:48.531888Z","iopub.status.idle":"2022-07-03T22:08:48.539055Z","shell.execute_reply.started":"2022-07-03T22:08:48.531846Z","shell.execute_reply":"2022-07-03T22:08:48.537946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #train data was in english so can only predict with\npreds = model.predict(test_text, verbose=1)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:08:50.651321Z","iopub.execute_input":"2022-07-03T22:08:50.652123Z","iopub.status.idle":"2022-07-03T22:08:52.350831Z","shell.execute_reply.started":"2022-07-03T22:08:50.652078Z","shell.execute_reply":"2022-07-03T22:08:52.349679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:06:08.556026Z","iopub.execute_input":"2022-07-03T22:06:08.556797Z","iopub.status.idle":"2022-07-03T22:06:08.631077Z","shell.execute_reply.started":"2022-07-03T22:06:08.556745Z","shell.execute_reply":"2022-07-03T22:06:08.630030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp2_df['score'] = preds[:, 0]\nsubmission_df = temp2_df[['id', 'comment_text','score']]\n\nsubmission_df.to_csv(\"submission.csv\", index=False)\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-07-03T22:09:03.091300Z","iopub.execute_input":"2022-07-03T22:09:03.091670Z","iopub.status.idle":"2022-07-03T22:09:03.112606Z","shell.execute_reply.started":"2022-07-03T22:09:03.091633Z","shell.execute_reply":"2022-07-03T22:09:03.111573Z"},"trusted":true},"execution_count":null,"outputs":[]}]}