{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"The dataset contains argumentative essays written by U.S students in grades 6-12. The essays were annotated by expert raters for elements commonly found in argumentative writing.\n\nTask: To predict the human annotations. You will first need to segment each essay into discrete rhetorical and argumentative elements (i.e., discourse elements) and then classify as one of 7 \"discourse types\". \n\nfor the text EDA please refer: https://www.kaggle.com/rachanabisht/evaluatingstudentwriting-complete-text-eda","metadata":{}},{"cell_type":"code","source":"# import libraries:\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\n\n\nfrom nltk.corpus import stopwords\nfrom tqdm.notebook import tqdm\n#import warnings\n#warnings.filterwarnings('ignore')\n\n#lib for model building:\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer\n\n\nfrom sklearn import model_selection, preprocessing, linear_model, naive_bayes, metrics, svm\n#from sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn import decomposition, ensemble\n\nimport xgboost, numpy, textblob, string\n\n\n#important lib for text processing\nimport os\nfrom wordcloud import WordCloud, STOPWORDS\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport nltk\nnltk.download(['punkt', 'wordnet'])\nnltk.download('stopwords')\nfrom nltk.tokenize import word_tokenize\nfrom nltk.tokenize import sent_tokenize\nfrom nltk.stem import WordNetLemmatizer\n\nimport re\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import RegexpTokenizer\nfrom sqlalchemy import create_engine  \npd.set_option('display.max_colwidth', None)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A. Getting the DATA:\n","metadata":{}},{"cell_type":"code","source":"base_path = '/kaggle/input/feedback-prize-2021/'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/feedback-prize-2021/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#A look at the annotated text_csv:\ndisplay(train_df.head(3))\ndisplay(train_df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets define the fetures and dependent variables:\ntrain_text = train_df[['discourse_text']]\ntrain_text.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# also define the dependent varaible as train_labels:\ntrain_labels = train_df[['discourse_type']]\ntrain_labels.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:01.042689Z","iopub.execute_input":"2022-03-15T15:20:01.043021Z","iopub.status.idle":"2022-03-15T15:20:01.050168Z","shell.execute_reply.started":"2022-03-15T15:20:01.042978Z","shell.execute_reply":"2022-03-15T15:20:01.049465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label = np.array(train_df.discourse_type)\ntrain_label","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:01.051438Z","iopub.execute_input":"2022-03-15T15:20:01.05167Z","iopub.status.idle":"2022-03-15T15:20:01.063577Z","shell.execute_reply.started":"2022-03-15T15:20:01.051644Z","shell.execute_reply":"2022-03-15T15:20:01.062731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# B. Data preprocessing:","metadata":{}},{"cell_type":"code","source":"# define a 'clean_text' function to process the text:\ndef clean_text(text, remove_stopwords=True, stem_words=False, lemma=True):\n    #text = str(text).lower().split()\n    text = str(text)\n    text = text.lower().split()\n    #remove stop words\n    if remove_stopwords:\n        stops = stopwords.words(\"english\")\n        text = [x for x in text if not x in stops]\n\n    \n    \n    text = ' '.join(text)\n    \n    text = re.sub(r\"[-()\\\"#/<>!@&;*:<>{}`'+=~%|.!?,_]\", \" \", text)\n    text = re.sub(r\"\\]\", \" \", text)\n    text = re.sub(r\"\\[\", \" \", text)\n    text = re.sub(r\"\\/\", \" \", text)\n    text = re.sub(r\"\\\\\", \" \", text)\n    text = re.sub(r\"\\'ve\", \" have \", text)\n    text = re.sub(r\"can't\", \"cannot \", text)\n    text = re.sub(r\"n't\", \" not \", text)\n    text = re.sub(r\"\\'re\", \" are \", text)\n    text = re.sub(r\"\\'d\", \" would \", text)\n    text = re.sub(r\"\\'ll\", \" will \", text)\n    text = re.sub(r\"  \", \" \", text)\n    text = re.sub(r\"   \", \" \", text)\n    text = re.sub(r\"   \", \" \", text)\n    text = re.sub(r\"0x00\", \"\", text)\n    \n    \n    if stem_words:\n        text = text.split()\n        stemmer = SnowballStemmer('english')\n        stem_words = [stemmer.stem(x) for x in text]\n        text = \" \".join(text)\n        \n    if lemma:\n        text = text.split()\n        lem = WordNetLemmatizer()\n        lemmatized = [lem.lemmatize(x, \"v\") for x in text]\n        text = \" \".join(text)\n        \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:01.064618Z","iopub.execute_input":"2022-03-15T15:20:01.065013Z","iopub.status.idle":"2022-03-15T15:20:01.078144Z","shell.execute_reply.started":"2022-03-15T15:20:01.064972Z","shell.execute_reply":"2022-03-15T15:20:01.077415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply clean_text function to text :\ntrain_text['cleaned_text'] = train_text.discourse_text.apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:01.079155Z","iopub.execute_input":"2022-03-15T15:20:01.079804Z","iopub.status.idle":"2022-03-15T15:20:59.254833Z","shell.execute_reply.started":"2022-03-15T15:20:01.079758Z","shell.execute_reply":"2022-03-15T15:20:59.254079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_text.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.257924Z","iopub.execute_input":"2022-03-15T15:20:59.258339Z","iopub.status.idle":"2022-03-15T15:20:59.269108Z","shell.execute_reply.started":"2022-03-15T15:20:59.258292Z","shell.execute_reply":"2022-03-15T15:20:59.268284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# C. Model ","metadata":{}},{"cell_type":"code","source":"# Defining X (feature) and y (target variables)\nX = train_text['cleaned_text'].values\ny =train_label","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.270208Z","iopub.execute_input":"2022-03-15T15:20:59.270419Z","iopub.status.idle":"2022-03-15T15:20:59.274781Z","shell.execute_reply.started":"2022-03-15T15:20:59.270389Z","shell.execute_reply":"2022-03-15T15:20:59.274128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.275949Z","iopub.execute_input":"2022-03-15T15:20:59.276412Z","iopub.status.idle":"2022-03-15T15:20:59.287783Z","shell.execute_reply.started":"2022-03-15T15:20:59.276381Z","shell.execute_reply":"2022-03-15T15:20:59.287149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.289145Z","iopub.execute_input":"2022-03-15T15:20:59.289617Z","iopub.status.idle":"2022-03-15T15:20:59.299389Z","shell.execute_reply.started":"2022-03-15T15:20:59.289574Z","shell.execute_reply":"2022-03-15T15:20:59.298744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# split the dataset into training and validation datasets (default splitsize == 0.25:)\nX_train, X_vtest, y_train, y_vtest = train_test_split(X, y)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.300733Z","iopub.execute_input":"2022-03-15T15:20:59.30113Z","iopub.status.idle":"2022-03-15T15:20:59.329211Z","shell.execute_reply.started":"2022-03-15T15:20:59.3011Z","shell.execute_reply":"2022-03-15T15:20:59.328465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label encode the target variable \nencoder = preprocessing.LabelEncoder()\ny_train = encoder.fit_transform(y_train)\ny_vtest = encoder.fit_transform(y_vtest)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.330377Z","iopub.execute_input":"2022-03-15T15:20:59.330714Z","iopub.status.idle":"2022-03-15T15:20:59.384548Z","shell.execute_reply.started":"2022-03-15T15:20:59.330686Z","shell.execute_reply":"2022-03-15T15:20:59.383896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:20:59.385592Z","iopub.execute_input":"2022-03-15T15:20:59.38592Z","iopub.status.idle":"2022-03-15T15:20:59.390582Z","shell.execute_reply.started":"2022-03-15T15:20:59.385893Z","shell.execute_reply":"2022-03-15T15:20:59.390025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate vectorizers and classifier\nvect = CountVectorizer()\ntfidf = TfidfTransformer()\nclf = RandomForestClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:21:35.479036Z","iopub.execute_input":"2022-03-15T15:21:35.479586Z","iopub.status.idle":"2022-03-15T15:21:35.484018Z","shell.execute_reply.started":"2022-03-15T15:21:35.479551Z","shell.execute_reply":"2022-03-15T15:21:35.483032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit transform CountVectorizer:\nX_train_counts = vect.fit_transform(X_train)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:21:41.203656Z","iopub.execute_input":"2022-03-15T15:21:41.20392Z","iopub.status.idle":"2022-03-15T15:21:44.424762Z","shell.execute_reply.started":"2022-03-15T15:21:41.203892Z","shell.execute_reply":"2022-03-15T15:21:44.423542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fit Transform TF-IDF vectorizer:\nX_train_tfidf = tfidf.fit_transform(X_train_counts)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:21:53.744204Z","iopub.execute_input":"2022-03-15T15:21:53.744506Z","iopub.status.idle":"2022-03-15T15:21:53.906835Z","shell.execute_reply.started":"2022-03-15T15:21:53.744475Z","shell.execute_reply":"2022-03-15T15:21:53.905729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train Random Forest classifier:\nmodel = clf.fit(X_train_tfidf, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:21:59.287842Z","iopub.execute_input":"2022-03-15T15:21:59.288142Z","iopub.status.idle":"2022-03-15T15:34:25.648282Z","shell.execute_reply.started":"2022-03-15T15:21:59.288112Z","shell.execute_reply":"2022-03-15T15:34:25.64764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict on validation data\nX_vtest_counts = vect.transform(X_vtest)\nX_vtest_tfidf = tfidf.transform(X_vtest_counts)\ny_pred = model.predict(X_vtest_tfidf)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:34:25.649499Z","iopub.execute_input":"2022-03-15T15:34:25.650193Z","iopub.status.idle":"2022-03-15T15:34:30.832271Z","shell.execute_reply.started":"2022-03-15T15:34:25.650157Z","shell.execute_reply":"2022-03-15T15:34:30.831308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define a function to display the results:\ndef display_results(y_vtest, y_vpred):\n    labels = np.unique(y_vpred)\n    confusion_mat = confusion_matrix(y_vtest, y_vpred, labels=labels)\n    accuracy = (y_vpred == y_vtest).mean()\n    print(\"Labels:\", labels)\n    print(\"Confusion Matrix:\\n\", confusion_mat)\n    print(\"Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:34:30.833465Z","iopub.execute_input":"2022-03-15T15:34:30.833687Z","iopub.status.idle":"2022-03-15T15:34:30.83941Z","shell.execute_reply.started":"2022-03-15T15:34:30.83366Z","shell.execute_reply":"2022-03-15T15:34:30.838218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_results(y_vtest, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:34:30.841039Z","iopub.execute_input":"2022-03-15T15:34:30.841342Z","iopub.status.idle":"2022-03-15T15:34:30.866257Z","shell.execute_reply.started":"2022-03-15T15:34:30.841312Z","shell.execute_reply":"2022-03-15T15:34:30.86512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n","metadata":{}},{"cell_type":"markdown","source":"# Preprocess the TEST data: \nWe have the following zip files:\n\ntest.zip - folder of individual .txt files, with each file containing the full text of an essay response in the test set\n\nsample_submission.csv - file in the required format for making predictions - note that if you are making multiple predictions for a document, submit multiple rows\n\ncredits: https://www.kaggle.com/nehapawar/tfidf-random-forest-classifier/comments","metadata":{}},{"cell_type":"code","source":"#process the test data into strings which are numbered by indices:\nTEST_PATH = base_path + 'test/'\n\ndef get_test_text(a_id):\n    a_file = f\"{TEST_PATH}/{a_id}.txt\"\n    with open(a_file, \"r\") as fp:\n        txt = fp.read()\n    return txt\n\ndef create_df_test():\n    test_ids = [f[:-4] for f in os.listdir(TEST_PATH)] #Remove the last 4 characters ('.txt') in the filenames such as '0FB0700DAF44.txt'.\n    test_data = []\n    for test_id in test_ids:\n        text = get_test_text(test_id)\n        sentences = nltk.sent_tokenize(text)\n        id_sentences = []\n        idx = 0 \n        for sentence in sentences:\n            id_sentence = []\n            words = sentence.split()\n            # I created this heuristic for mapping words in sentences to \"word indices\"\n            # This is not definitive and might have strong drawbacks and problems\n            for w in words:\n                id_sentence.append(idx)\n                idx+=1\n            id_sentences.append(id_sentence)\n        test_data += list(zip([test_id] * len(sentences), sentences, id_sentences))\n    df_test = pd.DataFrame(test_data, columns=['id', 'discourse_text', 'ids'])\n    return df_test","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:06.570712Z","iopub.execute_input":"2022-03-15T15:36:06.571007Z","iopub.status.idle":"2022-03-15T15:36:06.580486Z","shell.execute_reply.started":"2022-03-15T15:36:06.570977Z","shell.execute_reply":"2022-03-15T15:36:06.5798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = create_df_test()\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:07.429733Z","iopub.execute_input":"2022-03-15T15:36:07.430478Z","iopub.status.idle":"2022-03-15T15:36:07.493584Z","shell.execute_reply.started":"2022-03-15T15:36:07.430445Z","shell.execute_reply":"2022-03-15T15:36:07.493025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['predictionstring'] = df_test['ids'].apply(lambda x: ' '.join([str(i) for i in x]))\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:08.833432Z","iopub.execute_input":"2022-03-15T15:36:08.834091Z","iopub.status.idle":"2022-03-15T15:36:08.85236Z","shell.execute_reply.started":"2022-03-15T15:36:08.834056Z","shell.execute_reply":"2022-03-15T15:36:08.851425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.drop('ids', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:09.541558Z","iopub.execute_input":"2022-03-15T15:36:09.541845Z","iopub.status.idle":"2022-03-15T15:36:09.547547Z","shell.execute_reply.started":"2022-03-15T15:36:09.541807Z","shell.execute_reply":"2022-03-15T15:36:09.546728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:10.381228Z","iopub.execute_input":"2022-03-15T15:36:10.382058Z","iopub.status.idle":"2022-03-15T15:36:10.392974Z","shell.execute_reply.started":"2022-03-15T15:36:10.382Z","shell.execute_reply":"2022-03-15T15:36:10.392032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#apply clean_text function for text preprocessing:\ndf_test['cleaned_text'] = df_test.discourse_text.apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:11.261389Z","iopub.execute_input":"2022-03-15T15:36:11.262156Z","iopub.status.idle":"2022-03-15T15:36:11.319709Z","shell.execute_reply.started":"2022-03-15T15:36:11.262112Z","shell.execute_reply":"2022-03-15T15:36:11.318827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:12.041338Z","iopub.execute_input":"2022-03-15T15:36:12.041644Z","iopub.status.idle":"2022-03-15T15:36:12.052528Z","shell.execute_reply.started":"2022-03-15T15:36:12.041616Z","shell.execute_reply":"2022-03-15T15:36:12.051543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = df_test['cleaned_text'].values","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:12.885242Z","iopub.execute_input":"2022-03-15T15:36:12.88584Z","iopub.status.idle":"2022-03-15T15:36:12.890593Z","shell.execute_reply.started":"2022-03-15T15:36:12.885798Z","shell.execute_reply":"2022-03-15T15:36:12.890031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:13.917658Z","iopub.execute_input":"2022-03-15T15:36:13.918393Z","iopub.status.idle":"2022-03-15T15:36:13.925823Z","shell.execute_reply.started":"2022-03-15T15:36:13.918355Z","shell.execute_reply":"2022-03-15T15:36:13.924918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply vectorizers:\ntest_count = vect.transform(test)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:15.986925Z","iopub.execute_input":"2022-03-15T15:36:15.987366Z","iopub.status.idle":"2022-03-15T15:36:15.994899Z","shell.execute_reply.started":"2022-03-15T15:36:15.987334Z","shell.execute_reply":"2022-03-15T15:36:15.994316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tfidf_vec = tfidf.transform(test_count)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:17.577453Z","iopub.execute_input":"2022-03-15T15:36:17.577883Z","iopub.status.idle":"2022-03-15T15:36:17.583532Z","shell.execute_reply.started":"2022-03-15T15:36:17.577843Z","shell.execute_reply":"2022-03-15T15:36:17.582972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply the trained classifier to predict on test data\ny_final_pred = model.predict(test_tfidf_vec)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:36:23.981662Z","iopub.execute_input":"2022-03-15T15:36:23.982217Z","iopub.status.idle":"2022-03-15T15:36:24.041491Z","shell.execute_reply.started":"2022-03-15T15:36:23.982167Z","shell.execute_reply":"2022-03-15T15:36:24.040817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# submission:","metadata":{}},{"cell_type":"code","source":"submission_df = pd.DataFrame()\nsubmission_df['id'] = df_test['id']\nsubmission_df['class'] = y_final_pred# label of y_final_predict\nsubmission_df['predictionstring'] = df_test['predictionstring']","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:37:28.430182Z","iopub.execute_input":"2022-03-15T15:37:28.430771Z","iopub.status.idle":"2022-03-15T15:37:28.439252Z","shell.execute_reply.started":"2022-03-15T15:37:28.430719Z","shell.execute_reply":"2022-03-15T15:37:28.4384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T15:37:42.37825Z","iopub.execute_input":"2022-03-15T15:37:42.378529Z","iopub.status.idle":"2022-03-15T15:37:42.38829Z","shell.execute_reply.started":"2022-03-15T15:37:42.378499Z","shell.execute_reply":"2022-03-15T15:37:42.387577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df_2 = submission_df.copy()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-15T16:00:08.893219Z","iopub.execute_input":"2022-03-15T16:00:08.893532Z","iopub.status.idle":"2022-03-15T16:00:08.897949Z","shell.execute_reply.started":"2022-03-15T16:00:08.8935Z","shell.execute_reply":"2022-03-15T16:00:08.897335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reverse the process of LabelEncoder\nsubmission_df_2['class'] = encoder.inverse_transform(submission_df['class'])","metadata":{"execution":{"iopub.status.busy":"2022-03-15T16:01:54.897107Z","iopub.execute_input":"2022-03-15T16:01:54.897668Z","iopub.status.idle":"2022-03-15T16:01:54.902118Z","shell.execute_reply.started":"2022-03-15T16:01:54.897634Z","shell.execute_reply":"2022-03-15T16:01:54.901296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df_2.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-15T16:00:25.057381Z","iopub.execute_input":"2022-03-15T16:00:25.057705Z","iopub.status.idle":"2022-03-15T16:00:25.068717Z","shell.execute_reply.started":"2022-03-15T16:00:25.057671Z","shell.execute_reply":"2022-03-15T16:00:25.067745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df_2.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T16:01:14.008466Z","iopub.execute_input":"2022-03-15T16:01:14.00905Z","iopub.status.idle":"2022-03-15T16:01:14.014879Z","shell.execute_reply.started":"2022-03-15T16:01:14.009016Z","shell.execute_reply":"2022-03-15T16:01:14.014055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}