{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:00.518252Z","iopub.execute_input":"2022-07-17T19:51:00.518611Z","iopub.status.idle":"2022-07-17T19:51:00.550897Z","shell.execute_reply.started":"2022-07-17T19:51:00.518527Z","shell.execute_reply":"2022-07-17T19:51:00.549883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport re","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:00.552787Z","iopub.execute_input":"2022-07-17T19:51:00.553167Z","iopub.status.idle":"2022-07-17T19:51:01.561237Z","shell.execute_reply.started":"2022-07-17T19:51:00.553128Z","shell.execute_reply":"2022-07-17T19:51:01.560152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('../input/nlp-getting-started/train.csv',usecols=['id','text','target'])\ntest_data = pd.read_csv('../input/nlp-getting-started/test.csv',usecols=['id','text'])\nsample_data = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.564127Z","iopub.execute_input":"2022-07-17T19:51:01.564845Z","iopub.status.idle":"2022-07-17T19:51:01.627634Z","shell.execute_reply.started":"2022-07-17T19:51:01.564808Z","shell.execute_reply":"2022-07-17T19:51:01.626740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.columns)\nprint(test_data.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.630363Z","iopub.execute_input":"2022-07-17T19:51:01.630938Z","iopub.status.idle":"2022-07-17T19:51:01.637221Z","shell.execute_reply.started":"2022-07-17T19:51:01.630900Z","shell.execute_reply":"2022-07-17T19:51:01.635925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.641019Z","iopub.execute_input":"2022-07-17T19:51:01.641359Z","iopub.status.idle":"2022-07-17T19:51:01.660160Z","shell.execute_reply.started":"2022-07-17T19:51:01.641325Z","shell.execute_reply":"2022-07-17T19:51:01.658976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.661727Z","iopub.execute_input":"2022-07-17T19:51:01.662067Z","iopub.status.idle":"2022-07-17T19:51:01.672388Z","shell.execute_reply.started":"2022-07-17T19:51:01.662034Z","shell.execute_reply":"2022-07-17T19:51:01.671155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print multiple statments in same line\nfrom IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = 'all'","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.674227Z","iopub.execute_input":"2022-07-17T19:51:01.675019Z","iopub.status.idle":"2022-07-17T19:51:01.680674Z","shell.execute_reply.started":"2022-07-17T19:51:01.674982Z","shell.execute_reply":"2022-07-17T19:51:01.679526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape\ntest_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.682035Z","iopub.execute_input":"2022-07-17T19:51:01.682606Z","iopub.status.idle":"2022-07-17T19:51:01.696411Z","shell.execute_reply.started":"2022-07-17T19:51:01.682564Z","shell.execute_reply":"2022-07-17T19:51:01.695383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# take only columns that are needed for processing.\ntrain = train_data[['id','text', 'target']]\ntest = test_data[['id','text']]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.697668Z","iopub.execute_input":"2022-07-17T19:51:01.698534Z","iopub.status.idle":"2022-07-17T19:51:01.709970Z","shell.execute_reply.started":"2022-07-17T19:51:01.698486Z","shell.execute_reply":"2022-07-17T19:51:01.709140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check and remove nulls in test and train dataset\nprint('train isnull count \\n', train.isnull().sum())\nprint('\\n')\nprint('test is null count \\n', test.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.712478Z","iopub.execute_input":"2022-07-17T19:51:01.713180Z","iopub.status.idle":"2022-07-17T19:51:01.726340Z","shell.execute_reply.started":"2022-07-17T19:51:01.713142Z","shell.execute_reply":"2022-07-17T19:51:01.725191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import html\n\n# pre process data\ndef preprocess_text(sen):\n# Convert html entities to normal\n    sentence = html.unescape(sen)\n\n# Remove html tags\n    sentence = remove_tags(sentence)\n\n# Remove newline chars\n    sentence = remove_newlinechars(sentence)\n\n# Remove punctuations and numbers\n    sentence = re.sub('[^a-zA-Z]', ' ', sentence)\n\n# Convert to lowercase\n    sentence = sentence.lower()\n    return sentence\n\ndef remove_newlinechars(text):\n    return \" \".join(text.splitlines()) \n\ndef remove_tags(text):\n    TAG_RE = re.compile(r'<[^>]+>')\n    return TAG_RE.sub('', text)\n\ntrain['text'] = train['text'].apply(preprocess_text)\ntest['text'] = test['text'].apply(preprocess_text)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.731249Z","iopub.execute_input":"2022-07-17T19:51:01.732194Z","iopub.status.idle":"2022-07-17T19:51:01.888712Z","shell.execute_reply.started":"2022-07-17T19:51:01.732157Z","shell.execute_reply":"2022-07-17T19:51:01.887827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove stop words\n!pip install nlppreprocess\nfrom nlppreprocess import NLP\n\nnlp = NLP()\n\ntrain['text'] = train['text'].apply(nlp.process)\ntest['text'] = test['text'].apply(nlp.process)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:01.889942Z","iopub.execute_input":"2022-07-17T19:51:01.890282Z","iopub.status.idle":"2022-07-17T19:51:16.003719Z","shell.execute_reply.started":"2022-07-17T19:51:01.890247Z","shell.execute_reply":"2022-07-17T19:51:16.002588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install spacy\n!python -m spacy download en\nimport spacy\nen_model = spacy.load('en_core_web_sm', disable=['parser', 'ner'])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:16.008464Z","iopub.execute_input":"2022-07-17T19:51:16.011183Z","iopub.status.idle":"2022-07-17T19:51:44.151242Z","shell.execute_reply.started":"2022-07-17T19:51:16.011141Z","shell.execute_reply":"2022-07-17T19:51:44.150231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to lemmatize text\ndef lemmatization(texts):\n    output = []\n    for i in texts:\n        s = [token.lemma_ for token in en_model(i)]\n        output.append(' '.join(s))\n    return output\n\n# Applying the lemmatization function to both test and training datasets\ntrain['text'] = lemmatization(train['text'])\ntest['text'] = lemmatization(test['text'])","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:51:44.152750Z","iopub.execute_input":"2022-07-17T19:51:44.153873Z","iopub.status.idle":"2022-07-17T19:52:17.225775Z","shell.execute_reply.started":"2022-07-17T19:51:44.153832Z","shell.execute_reply":"2022-07-17T19:52:17.224801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('unique label train', train.target.nunique()) \n\nprint(\"max len of train text \",max([len(x.split()) for x in train.text])) \nprint(\"max len of test text \",max([len(x.split()) for x in test.text])) \n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:17.227308Z","iopub.execute_input":"2022-07-17T19:52:17.227682Z","iopub.status.idle":"2022-07-17T19:52:17.248557Z","shell.execute_reply.started":"2022-07-17T19:52:17.227646Z","shell.execute_reply":"2022-07-17T19:52:17.247530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_len = 30\nnclass = 2","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:17.250199Z","iopub.execute_input":"2022-07-17T19:52:17.250617Z","iopub.status.idle":"2022-07-17T19:52:17.259320Z","shell.execute_reply.started":"2022-07-17T19:52:17.250567Z","shell.execute_reply":"2022-07-17T19:52:17.258294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(\n    train['text'], \n    train['target'], \n    test_size = 0.2, \n    random_state = 1, \n    stratify=train['target']\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:17.260580Z","iopub.execute_input":"2022-07-17T19:52:17.260877Z","iopub.status.idle":"2022-07-17T19:52:17.276155Z","shell.execute_reply.started":"2022-07-17T19:52:17.260854Z","shell.execute_reply":"2022-07-17T19:52:17.275107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.notebook import tqdm_notebook\ntqdm_notebook.pandas()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:17.277293Z","iopub.execute_input":"2022-07-17T19:52:17.277717Z","iopub.status.idle":"2022-07-17T19:52:17.285725Z","shell.execute_reply.started":"2022-07-17T19:52:17.277664Z","shell.execute_reply":"2022-07-17T19:52:17.284167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer,TFBertModel, TFBertForSequenceClassification\n\nimport tensorflow as tf\ntf.config.experimental.list_physical_devices('GPU')\n\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom tensorflow.keras.initializers import TruncatedNormal\nfrom tensorflow.keras.losses import CategoricalCrossentropy,BinaryCrossentropy\nfrom tensorflow.keras.metrics import CategoricalAccuracy,BinaryAccuracy\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.utils import plot_model\nfrom tensorflow.keras.layers import Input, Dense","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:17.287516Z","iopub.execute_input":"2022-07-17T19:52:17.288343Z","iopub.status.idle":"2022-07-17T19:52:24.020776Z","shell.execute_reply.started":"2022-07-17T19:52:17.288303Z","shell.execute_reply":"2022-07-17T19:52:24.019696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bert-base-uncased -- use this insted of bert-large-uncased\ntokenizer = AutoTokenizer.from_pretrained('bert-base-uncased')\n#bert = TFBertModel.from_pretrained('bert-base-uncased')\nmodel = TFBertModel.from_pretrained('bert-base-uncased', trainable=True, num_labels=nclass)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:52:24.022151Z","iopub.execute_input":"2022-07-17T19:52:24.023245Z","iopub.status.idle":"2022-07-17T19:53:02.855359Z","shell.execute_reply.started":"2022-07-17T19:52:24.023202Z","shell.execute_reply":"2022-07-17T19:53:02.854360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check if tokenizer works\nt = tokenizer('Happy learning and keep kaggling &*&*&&')\nt","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:02.856735Z","iopub.execute_input":"2022-07-17T19:53:02.857690Z","iopub.status.idle":"2022-07-17T19:53:02.879669Z","shell.execute_reply.started":"2022-07-17T19:53:02.857651Z","shell.execute_reply":"2022-07-17T19:53:02.878661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batch_encode(X, tokenizer):\n    return tokenizer.batch_encode_plus(\n    X.tolist(),\n    add_special_tokens=True,\n    max_length=max_len, \n    truncation=True,\n    padding=True, \n     # add [CLS] and [SEP] tokens\n    return_attention_mask=True,\n    return_token_type_ids=False, # not needed for this type of ML task\n    #pad_to_max_length=True, # add 0 pad tokens to the sequences less than max_length\n    return_tensors='tf',\n    verbose = True\n)\n\nX_train = batch_encode(X_train, tokenizer)\nX_val = batch_encode(X_val, tokenizer)\nX_test = batch_encode(test.text.values, tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:02.882756Z","iopub.execute_input":"2022-07-17T19:53:02.883082Z","iopub.status.idle":"2022-07-17T19:53:04.491630Z","shell.execute_reply.started":"2022-07-17T19:53:02.883040Z","shell.execute_reply":"2022-07-17T19:53:04.490615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_ids\")\nattention_mask = Input(shape=(max_len,), dtype=tf.int32, name=\"attention_mask\")\n\nembeddings = model([input_ids, attention_mask])[1] #(0 is the last hidden states,1 means pooler_output)\n\nout = tf.keras.layers.Dropout(0.1)(embeddings)\n\nout = Dense(128, activation='relu')(out)\nout = tf.keras.layers.Dropout(0.1)(out)\nout = Dense(32,activation = 'relu')(out)\n\ny = Dense(1,activation = 'sigmoid')(out)\n    \nmodel = tf.keras.Model(inputs=[input_ids, attention_mask], outputs=y)\n\noptimizer = Adam(\n    learning_rate=6e-05, # this learning rate is for bert model.\n    epsilon=1e-08,\n    decay=0.01,\n    clipnorm=1.0)\n\n# Set loss and metrics\nloss = BinaryCrossentropy(from_logits = True)\nmetric = BinaryAccuracy('accuracy')\n\n# Compile the model\nmodel.compile(\n    optimizer = optimizer,\n    loss = loss, \n    metrics = metric)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:04.493161Z","iopub.execute_input":"2022-07-17T19:53:04.493568Z","iopub.status.idle":"2022-07-17T19:53:12.184154Z","shell.execute_reply.started":"2022-07-17T19:53:04.493526Z","shell.execute_reply":"2022-07-17T19:53:12.183117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:12.185599Z","iopub.execute_input":"2022-07-17T19:53:12.186137Z","iopub.status.idle":"2022-07-17T19:53:12.206594Z","shell.execute_reply.started":"2022-07-17T19:53:12.186101Z","shell.execute_reply":"2022-07-17T19:53:12.205539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_model(model, show_shapes = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:12.208070Z","iopub.execute_input":"2022-07-17T19:53:12.208556Z","iopub.status.idle":"2022-07-17T19:53:13.431885Z","shell.execute_reply.started":"2022-07-17T19:53:12.208518Z","shell.execute_reply":"2022-07-17T19:53:13.430284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit the model\nfinal = model.fit(\n    x=X_train.values(),\n    y=y_train,\n    validation_data=(X_val.values(), y_val),\n    epochs=4,\n    batch_size=32\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:53:13.433240Z","iopub.execute_input":"2022-07-17T19:53:13.433825Z","iopub.status.idle":"2022-07-17T19:56:52.002650Z","shell.execute_reply.started":"2022-07-17T19:53:13.433789Z","shell.execute_reply":"2022-07-17T19:56:52.001653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# VISUALIZATION OF LOSS AND ACCURACY CURVE:¶","metadata":{}},{"cell_type":"code","source":"def visual_accuracy_and_loss(final):\n    acc = final.history['accuracy']\n    loss = final.history['loss']\n    epochs_plot = np.arange(1, len(loss) + 1)\n    plt.clf()\n    plt.plot(epochs_plot, acc, 'r', label='Accuracy')\n    plt.plot(epochs_plot, loss, 'b:', label='Loss')\n    plt.title('VISUALIZATION OF LOSS AND ACCURACY CURVE')\n    plt.xlabel('Epochs')\n    plt.legend()\n    plt.show()\n\nvisual_accuracy_and_loss(final)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:52.004748Z","iopub.execute_input":"2022-07-17T19:56:52.005117Z","iopub.status.idle":"2022-07-17T19:56:52.232310Z","shell.execute_reply.started":"2022-07-17T19:56:52.005080Z","shell.execute_reply":"2022-07-17T19:56:52.231419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the loss and accuracy curves  \n\n#Diffining Figure\nf = plt.figure(figsize=(20,7))\n\n#Adding Subplot 1 (For Accuracy)\nf.add_subplot(121)\n\nplt.plot(final.epoch,final.history['accuracy'],label = \"accuracy\") # Accuracy curve \n\n\nplt.title(\"Accuracy Curve\",fontsize=18)\nplt.xlabel(\"Epochs\",fontsize=15)\nplt.ylabel(\"Accuracy\",fontsize=15)\nplt.grid(alpha=0.3)\nplt.legend()\n\n#Adding Subplot 1 (For Loss)\nf.add_subplot(122)\n\nplt.plot(final.epoch,final.history['loss'],label=\"loss\") # Loss curve \n\n\nplt.title(\"Loss Curve\",fontsize=18)\nplt.xlabel(\"Epochs\",fontsize=15)\nplt.ylabel(\"Loss\",fontsize=15)\nplt.grid(alpha=0.3)\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:52.233998Z","iopub.execute_input":"2022-07-17T19:56:52.234345Z","iopub.status.idle":"2022-07-17T19:56:52.668307Z","shell.execute_reply.started":"2022-07-17T19:56:52.234308Z","shell.execute_reply":"2022-07-17T19:56:52.667169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = final.history\nhistory_dict.keys()\n\nacc = history_dict['accuracy']\nval_acc = history_dict['val_accuracy']\nloss = history_dict['loss']\nval_loss = history_dict['val_loss']\n\nepochs = range(1, len(acc) + 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:52.674931Z","iopub.execute_input":"2022-07-17T19:56:52.675234Z","iopub.status.idle":"2022-07-17T19:56:52.683316Z","shell.execute_reply.started":"2022-07-17T19:56:52.675205Z","shell.execute_reply":"2022-07-17T19:56:52.681987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:52.684911Z","iopub.execute_input":"2022-07-17T19:56:52.686057Z","iopub.status.idle":"2022-07-17T19:56:52.900435Z","shell.execute_reply.started":"2022-07-17T19:56:52.686017Z","shell.execute_reply":"2022-07-17T19:56:52.899512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.clf()   # clear figure\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:52.901961Z","iopub.execute_input":"2022-07-17T19:56:52.902324Z","iopub.status.idle":"2022-07-17T19:56:53.117419Z","shell.execute_reply.started":"2022-07-17T19:56:52.902288Z","shell.execute_reply":"2022-07-17T19:56:53.116533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predicted = model.predict(X_test.values())\npredicted.shape\n\ny_predicted = np.where(predicted > 0.5, 1, 0)\ny_predicted = y_predicted.reshape(1, len(y_predicted))\ny_predicted = y_predicted[0,:]\ny_predicted = y_predicted.tolist()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:56:53.118586Z","iopub.execute_input":"2022-07-17T19:56:53.118927Z","iopub.status.idle":"2022-07-17T19:57:01.956505Z","shell.execute_reply.started":"2022-07-17T19:56:53.118886Z","shell.execute_reply":"2022-07-17T19:57:01.955478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data['id'] = test['id']\nsample_data['target'] = y_predicted\nsample_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:57:01.957762Z","iopub.execute_input":"2022-07-17T19:57:01.958131Z","iopub.status.idle":"2022-07-17T19:57:01.974541Z","shell.execute_reply.started":"2022-07-17T19:57:01.958093Z","shell.execute_reply":"2022-07-17T19:57:01.973135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data.groupby('target').size()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:57:01.978225Z","iopub.execute_input":"2022-07-17T19:57:01.978485Z","iopub.status.idle":"2022-07-17T19:57:01.988483Z","shell.execute_reply.started":"2022-07-17T19:57:01.978461Z","shell.execute_reply":"2022-07-17T19:57:01.987462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T19:57:01.990396Z","iopub.execute_input":"2022-07-17T19:57:01.991241Z","iopub.status.idle":"2022-07-17T19:57:02.004335Z","shell.execute_reply.started":"2022-07-17T19:57:01.991206Z","shell.execute_reply":"2022-07-17T19:57:02.003332Z"},"trusted":true},"execution_count":null,"outputs":[]}]}