{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T17:43:21.173137Z","iopub.execute_input":"2022-08-11T17:43:21.173896Z","iopub.status.idle":"2022-08-11T17:43:21.184870Z","shell.execute_reply.started":"2022-08-11T17:43:21.173859Z","shell.execute_reply":"2022-08-11T17:43:21.183364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk\nnltk.download('omw-1.4')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.186599Z","iopub.execute_input":"2022-08-11T17:43:21.187190Z","iopub.status.idle":"2022-08-11T17:43:21.254798Z","shell.execute_reply.started":"2022-08-11T17:43:21.187153Z","shell.execute_reply":"2022-08-11T17:43:21.253827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hello world\nI have clearly explained the preprocessing steps in a previous notebook\nHere is the link if you have some trouble following the code for data processing https://www.kaggle.com/code/ajayuser/nlp-with-disaster-tweets-sklearn\n\nIf you like to see image preprocessing techniques in python, go checkout https://www.kaggle.com/code/ajayuser/facial-key-points-detection-preprocessing-guide\n\nToday the main goal is to train a TENSORFLOW MODEL\n\nWe will build a tesorflow model with attention mechanism for text classification\n\nIf you like to see a simple Bi-directional RNN, checkout: https://www.kaggle.com/code/ajayuser/nlp-classif-tensorflow-bi-directional-lstm\n\nIf you like to see a logistic regression model checkout : https://www.kaggle.com/code/ajayuser/nlp-classif-tensorflow-logistic-regression\n\nAs always please post your suggestions, reviews, recommendations in the comments section","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:35:10.010274Z","iopub.execute_input":"2022-08-11T06:35:10.01075Z","iopub.status.idle":"2022-08-11T06:35:10.019948Z","shell.execute_reply.started":"2022-08-11T06:35:10.010716Z","shell.execute_reply":"2022-08-11T06:35:10.018231Z"}}},{"cell_type":"code","source":"import re\nimport string\nimport tensorflow as tf\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.stem import PorterStemmer\nfrom nltk.stem import WordNetLemmatizer\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.256139Z","iopub.execute_input":"2022-08-11T17:43:21.257096Z","iopub.status.idle":"2022-08-11T17:43:21.263893Z","shell.execute_reply.started":"2022-08-11T17:43:21.257060Z","shell.execute_reply":"2022-08-11T17:43:21.262653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')\nsample_submission_df = pd.read_csv('/kaggle/input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.265758Z","iopub.execute_input":"2022-08-11T17:43:21.266800Z","iopub.status.idle":"2022-08-11T17:43:21.320307Z","shell.execute_reply.started":"2022-08-11T17:43:21.266755Z","shell.execute_reply":"2022-08-11T17:43:21.318083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.325075Z","iopub.execute_input":"2022-08-11T17:43:21.327393Z","iopub.status.idle":"2022-08-11T17:43:21.345139Z","shell.execute_reply.started":"2022-08-11T17:43:21.327352Z","shell.execute_reply":"2022-08-11T17:43:21.344231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.base import BaseEstimator, TransformerMixin\n\nclass textTransformer(TransformerMixin):\n    def __init__(self):\n        self.lemmatizer = WordNetLemmatizer()\n        self.stemmer = PorterStemmer()\n        self.english_stop_words = stopwords.words('english')\n    \n    def remove_html(self, text):\n        return re.sub(pattern=r'<\\w+/?>', repl='',string=text)\n    \n    def remove_mentioning(self, text):\n        return re.sub(pattern=r'@[^\\s]*', repl='',string=text)\n\n    def remove_url(self, text):\n        return re.sub(pattern=r'http[^\\s]*', repl='', string=text)\n    \n    def remove_punctuation(self, text):\n        return re.sub(pattern=r\"[{0}]\".format(string.punctuation),repl='',string=text)\n    \n    def remove_numbers(self, text):\n        return re.sub(pattern=r'\\d+', repl='', string=text)\n    \n    def remove_stop_words(self, text):\n        return ' '.join([word for word in word_tokenize(text) if word not in self.english_stop_words])\n    \n    def stemming(self, text):\n        return ' '.join([self.stemmer.stem(word) for word in word_tokenize(text)])\n    \n    def lemmatize(self, text, pos='v'):\n        return ' '.join([self.lemmatizer.lemmatize(word, pos=pos) for word in word_tokenize(text)])\n    \n    def fit(self,X):\n        return self\n    \n    def transform(self,X):\n        \n\n        X['processed_text']= (X['text']\n                              .apply(lambda text: text.lower())\n                              .apply(self.remove_html)\n                              .apply(self.remove_url)\n                              .apply(self.remove_mentioning)\n                              .apply(self.remove_punctuation)\n                              .apply(self.remove_numbers)\n                              .apply(self.remove_stop_words)\n                             )\n                           \n    \n            \n        X['lemmatized_text']= X['processed_text'].apply(self.lemmatize)\n        \n        X['stemmed_text']= X['processed_text'].apply(self.stemming)\n\n        \n        return X\n        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.349355Z","iopub.execute_input":"2022-08-11T17:43:21.351608Z","iopub.status.idle":"2022-08-11T17:43:21.369472Z","shell.execute_reply.started":"2022-08-11T17:43:21.351571Z","shell.execute_reply":"2022-08-11T17:43:21.368486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process Training data","metadata":{}},{"cell_type":"code","source":"%time\n# APPLY THE TRANSFORMATION\ntransformer = textTransformer()\n\nX = transformer.fit_transform(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:21.374142Z","iopub.execute_input":"2022-08-11T17:43:21.377055Z","iopub.status.idle":"2022-08-11T17:43:26.533877Z","shell.execute_reply.started":"2022-08-11T17:43:21.377018Z","shell.execute_reply":"2022-08-11T17:43:26.532897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CHECK THE DATA\nX[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:26.535513Z","iopub.execute_input":"2022-08-11T17:43:26.535907Z","iopub.status.idle":"2022-08-11T17:43:26.553660Z","shell.execute_reply.started":"2022-08-11T17:43:26.535867Z","shell.execute_reply":"2022-08-11T17:43:26.552009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# APPLY THE TRANSFORMATION TO TEST DATA\ntransformer = textTransformer()\n\nX_test = transformer.transform(test_df)\n\nX_test[['text','processed_text','lemmatized_text','stemmed_text']]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:26.555302Z","iopub.execute_input":"2022-08-11T17:43:26.555954Z","iopub.status.idle":"2022-08-11T17:43:28.553283Z","shell.execute_reply.started":"2022-08-11T17:43:26.555917Z","shell.execute_reply":"2022-08-11T17:43:28.552090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-11T06:44:43.877919Z","iopub.execute_input":"2022-08-11T06:44:43.878408Z","iopub.status.idle":"2022-08-11T06:44:43.883607Z","shell.execute_reply.started":"2022-08-11T06:44:43.878366Z","shell.execute_reply":"2022-08-11T06:44:43.882328Z"}}},{"cell_type":"code","source":"# Create a dataset\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((X['lemmatized_text'],train_df['target']))\ntest_ds = tf.data.Dataset.from_tensor_slices((X_test['lemmatized_text']))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:28.554987Z","iopub.execute_input":"2022-08-11T17:43:28.556488Z","iopub.status.idle":"2022-08-11T17:43:28.569545Z","shell.execute_reply.started":"2022-08-11T17:43:28.556449Z","shell.execute_reply":"2022-08-11T17:43:28.568472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = train_ds.batch(64).prefetch(tf.data.AUTOTUNE)\ntest_ds = test_ds.batch(64).prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:28.571004Z","iopub.execute_input":"2022-08-11T17:43:28.571381Z","iopub.status.idle":"2022-08-11T17:43:28.578871Z","shell.execute_reply.started":"2022-08-11T17:43:28.571346Z","shell.execute_reply":"2022-08-11T17:43:28.577564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_batch, label_batch = next(iter(train_ds))\n\nfor text, label in zip(text_batch,label_batch):\n    print(text)\n    print('label',label)\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:28.580515Z","iopub.execute_input":"2022-08-11T17:43:28.581104Z","iopub.status.idle":"2022-08-11T17:43:28.595262Z","shell.execute_reply.started":"2022-08-11T17:43:28.581017Z","shell.execute_reply":"2022-08-11T17:43:28.594420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Vocabulary","metadata":{}},{"cell_type":"code","source":"# lets find out the sequence length of different rows\n# create a histogram using the seq-len\n# and decide the maximum token-length \n# X[['text','processed_text','lemmatized_text','stemmed_text']]\nseq_len = []\nfor text in X['lemmatized_text']:\n    seq_len.append(len(word_tokenize(text)))\n\n\nplt.figure(figsize=(10,6))\nplt.subplot(121)\nplt.hist(seq_len);\nplt.subplot(122)\nsns.histplot(seq_len, kde=True);","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:28.602717Z","iopub.execute_input":"2022-08-11T17:43:28.603496Z","iopub.status.idle":"2022-08-11T17:43:29.722054Z","shell.execute_reply.started":"2022-08-11T17:43:28.603461Z","shell.execute_reply":"2022-08-11T17:43:29.721119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_TOKENS = 10000\nMAX_SEQ_LEN = 16\nEMBED_DIMS = 128\nDENSE_UNITS = 512","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.723668Z","iopub.execute_input":"2022-08-11T17:43:29.724051Z","iopub.status.idle":"2022-08-11T17:43:29.729287Z","shell.execute_reply.started":"2022-08-11T17:43:29.724013Z","shell.execute_reply":"2022-08-11T17:43:29.728190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# STANDARDIZE TOKENIZE VECTORIZE\ntext_vectorizer = tf.keras.layers.TextVectorization(\n    max_tokens=MAX_TOKENS,\n    output_mode='int',\n    output_sequence_length=MAX_SEQ_LEN\n)\n\n\n# ADAPT\n\ntrain_text = train_ds.map(lambda x,y:x, num_parallel_calls=tf.data.AUTOTUNE)\n\ntext_vectorizer.adapt(train_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.731103Z","iopub.execute_input":"2022-08-11T17:43:29.731535Z","iopub.status.idle":"2022-08-11T17:43:29.928568Z","shell.execute_reply.started":"2022-08-11T17:43:29.731498Z","shell.execute_reply":"2022-08-11T17:43:29.927569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets check our vocab\ntext_vectorizer.get_vocabulary()[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.930440Z","iopub.execute_input":"2022-08-11T17:43:29.930831Z","iopub.status.idle":"2022-08-11T17:43:29.959586Z","shell.execute_reply.started":"2022-08-11T17:43:29.930794Z","shell.execute_reply":"2022-08-11T17:43:29.958424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_vectorizer.vocabulary_size()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.961141Z","iopub.execute_input":"2022-08-11T17:43:29.961619Z","iopub.status.idle":"2022-08-11T17:43:29.968500Z","shell.execute_reply.started":"2022-08-11T17:43:29.961570Z","shell.execute_reply":"2022-08-11T17:43:29.967503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_text_tokens = text_vectorizer(text_batch)\nprint('Shape of text batch', text_batch.shape)\nprint('Shape of token batch', sample_text_tokens.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.970038Z","iopub.execute_input":"2022-08-11T17:43:29.970726Z","iopub.status.idle":"2022-08-11T17:43:29.984652Z","shell.execute_reply.started":"2022-08-11T17:43:29.970690Z","shell.execute_reply":"2022-08-11T17:43:29.983582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attention","metadata":{}},{"cell_type":"code","source":"class Attention(tf.keras.layers.Layer):\n    def __init__(self, units):\n        super(Attention, self).__init__()\n        self.dq = tf.keras.layers.Dense(units, use_bias=False)        \n        self.dv = tf.keras.layers.Dense(units, use_bias=False)   \n \n\n        self.attention = tf.keras.layers.AdditiveAttention()\n    \n    def call(self, query, value, mask):\n        q = self.dq(query)\n        v = self.dv(value)\n\n        attention, weights = self.attention(inputs=[q,v],\n                                            mask=[mask['query'], mask['value']],\n                                            return_attention_scores=True)\n        \n        \n        return attention, weights","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.986172Z","iopub.execute_input":"2022-08-11T17:43:29.986556Z","iopub.status.idle":"2022-08-11T17:43:29.993832Z","shell.execute_reply.started":"2022-08-11T17:43:29.986522Z","shell.execute_reply":"2022-08-11T17:43:29.992658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets try the attention\n\nx = tf.random.normal(shape=(64,16,256))  # [batch, seq, dims]\nattention = Attention(units=256)\n\nmask = {\n    'query': tf.cast(tf.ones(shape=(64,16)), dtype=tf.bool),\n    'value': tf.cast(tf.ones(shape=(64,16)), dtype=tf.bool)\n}\n\nvector, weights = attention(x,x,mask)\n\nprint('Shape of query : ',x.shape)\nprint('Shape of attention vector : ',vector.shape)\nprint('Shape of attention weights : ',weights.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:29.995645Z","iopub.execute_input":"2022-08-11T17:43:29.996000Z","iopub.status.idle":"2022-08-11T17:43:30.018902Z","shell.execute_reply.started":"2022-08-11T17:43:29.995967Z","shell.execute_reply":"2022-08-11T17:43:30.017747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Text Encoder","metadata":{}},{"cell_type":"code","source":"class Encoder(tf.keras.layers.Layer):\n    def __init__(self, input_vocab_size, dims, units, output_units=1):\n        super(Encoder, self).__init__()\n        self.embedding = tf.keras.layers.Embedding(input_vocab_size, dims)\n        self.gru = tf.keras.layers.GRU(units)\n        self.attention = Attention(units)\n        self.dense = tf.keras.layers.Dense(units, activation='tanh')\n        self.final = tf.keras.layers.Dense(output_units)\n        \n    def create_mask(self, tokens):\n        mask = {\n            'query' : tf.cast( tf.not_equal(tokens,0) ,dtype=tf.bool),\n            'value' : tf.cast( tf.not_equal(tokens,0) ,dtype=tf.bool)\n        }\n        return mask\n        \n    def call(self, text_tokens):\n        #  create mask\n        mask = self.create_mask(text_tokens)\n        \n        # [batch, seq] --> [batch, seq, dims]\n        x = self.embedding(text_tokens)\n        \n        # Self attention        \n        #[batch, seq, units],[batch, seq, seq]\n        attention_vector, attention_weights = self.attention(x,x,mask)\n        \n        # RNN\n        # [batch, seq, units] --> [batch, units]\n        rnn_out = self.gru(attention_vector)\n    \n        # [batch, units] \n        dense = self.dense(rnn_out)\n        # final output [batch, output_units]\n        final_output = self.final(dense)\n        \n        return final_output","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:30.020657Z","iopub.execute_input":"2022-08-11T17:43:30.021032Z","iopub.status.idle":"2022-08-11T17:43:30.030182Z","shell.execute_reply.started":"2022-08-11T17:43:30.020997Z","shell.execute_reply":"2022-08-11T17:43:30.028882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets try attention\n\nencoder = Encoder(MAX_TOKENS, EMBED_DIMS, DENSE_UNITS)\noutput_logits = encoder(sample_text_tokens)\nprint('Shape of input tokens', sample_text_tokens.shape)\nprint('Shape of output logits', output_logits.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:30.032138Z","iopub.execute_input":"2022-08-11T17:43:30.032702Z","iopub.status.idle":"2022-08-11T17:43:30.175766Z","shell.execute_reply.started":"2022-08-11T17:43:30.032668Z","shell.execute_reply":"2022-08-11T17:43:30.174731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T07:01:10.026628Z","iopub.execute_input":"2022-08-11T07:01:10.027847Z","iopub.status.idle":"2022-08-11T07:01:10.098895Z","shell.execute_reply.started":"2022-08-11T07:01:10.027795Z","shell.execute_reply":"2022-08-11T07:01:10.097526Z"}}},{"cell_type":"code","source":"encoder = Encoder(MAX_TOKENS, EMBED_DIMS, DENSE_UNITS)\n\nmodel = tf.keras.Sequential([\n    text_vectorizer,\n    encoder\n])\n\n\nmodel.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n              optimizer=tf.keras.optimizers.Adam(1e-4),\n              metrics=['accuracy'])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:30.178320Z","iopub.execute_input":"2022-08-11T17:43:30.179473Z","iopub.status.idle":"2022-08-11T17:43:31.018148Z","shell.execute_reply.started":"2022-08-11T17:43:30.179433Z","shell.execute_reply":"2022-08-11T17:43:31.017138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_ds, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:31.019850Z","iopub.execute_input":"2022-08-11T17:43:31.020245Z","iopub.status.idle":"2022-08-11T17:43:55.091815Z","shell.execute_reply.started":"2022-08-11T17:43:31.020187Z","shell.execute_reply":"2022-08-11T17:43:55.090912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction and submission","metadata":{}},{"cell_type":"code","source":"preds = model.predict(test_ds)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:55.093377Z","iopub.execute_input":"2022-08-11T17:43:55.093739Z","iopub.status.idle":"2022-08-11T17:43:56.811625Z","shell.execute_reply.started":"2022-08-11T17:43:55.093703Z","shell.execute_reply":"2022-08-11T17:43:56.810615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0\n\nprediction_df = pd.DataFrame(data={'text':test_df['text'],'label':np.where(preds<threshold,0,1).ravel()})\n\nprediction_df.head(25)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:56.813229Z","iopub.execute_input":"2022-08-11T17:43:56.813587Z","iopub.status.idle":"2022-08-11T17:43:56.828925Z","shell.execute_reply.started":"2022-08-11T17:43:56.813550Z","shell.execute_reply":"2022-08-11T17:43:56.827788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Submission","metadata":{}},{"cell_type":"code","source":"sample_submission_df['target']=np.where(preds<threshold,0,1).ravel()\nsample_submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:56.830760Z","iopub.execute_input":"2022-08-11T17:43:56.831392Z","iopub.status.idle":"2022-08-11T17:43:56.845369Z","shell.execute_reply.started":"2022-08-11T17:43:56.831348Z","shell.execute_reply":"2022-08-11T17:43:56.844375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:43:56.846819Z","iopub.execute_input":"2022-08-11T17:43:56.847278Z","iopub.status.idle":"2022-08-11T17:43:56.858567Z","shell.execute_reply.started":"2022-08-11T17:43:56.847239Z","shell.execute_reply":"2022-08-11T17:43:56.857659Z"},"trusted":true},"execution_count":null,"outputs":[]}]}