{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hi, everybody.\n\n### This notebook aim to explain what is the goal of the competition and some skills for begginers.","metadata":{}},{"cell_type":"markdown","source":"# Load train file","metadata":{}},{"cell_type":"code","source":"import pandas as pd\ndf_effective_args = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ndf_effective_args.head(3)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:10.743469Z","iopub.execute_input":"2022-08-04T08:27:10.743861Z","iopub.status.idle":"2022-08-04T08:27:10.921754Z","shell.execute_reply.started":"2022-08-04T08:27:10.743825Z","shell.execute_reply":"2022-08-04T08:27:10.920919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Competition","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure()\ndf_effective_args['discourse_effectiveness'].value_counts().plot.pie()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:10.923117Z","iopub.execute_input":"2022-08-04T08:27:10.923384Z","iopub.status.idle":"2022-08-04T08:27:11.001210Z","shell.execute_reply.started":"2022-08-04T08:27:10.923361Z","shell.execute_reply":"2022-08-04T08:27:10.999841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this competition you have to predict the discourse effectiveness with information from 'discourse_id' 'essay_id' 'discourse_text' 'discourse_type'","metadata":{}},{"cell_type":"markdown","source":"# First model : Basic approch (TF IDF + Logistic Regression )","metadata":{}},{"cell_type":"markdown","source":"## import","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport numpy as np\nimport sklearn\nimport ast\nfrom scipy import stats\n\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\n\nfrom nltk.corpus import wordnet\nfrom nltk.corpus import stopwords\nnltk.data.path.append('/kaggle/input/corporafolder')","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.003380Z","iopub.execute_input":"2022-08-04T08:27:11.004458Z","iopub.status.idle":"2022-08-04T08:27:11.011164Z","shell.execute_reply.started":"2022-08-04T08:27:11.004420Z","shell.execute_reply":"2022-08-04T08:27:11.010068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Note : I had some trouble with tf_idf approch due to omw-1.4 library.\nI had to download it separately, then adding to my current repo kaggle.\n\nyou will find .zip folder there : https://www.nltk.org/nltk_data/","metadata":{}},{"cell_type":"markdown","source":"## Cleaning","metadata":{}},{"cell_type":"markdown","source":"Load new train file without duplicates. This new train file is based on https://www.kaggle.com/code/iamleonie/feedback-prize-eda-starter-for-beginners observations","metadata":{}},{"cell_type":"code","source":"df_effective_args = pd.read_csv('../input/train-light-csv/train_light.csv',encoding=\"utf-8\",header=(0))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.012901Z","iopub.execute_input":"2022-08-04T08:27:11.013598Z","iopub.status.idle":"2022-08-04T08:27:11.123072Z","shell.execute_reply.started":"2022-08-04T08:27:11.013557Z","shell.execute_reply":"2022-08-04T08:27:11.122118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_token(string):\n  tokenizer = nltk.RegexpTokenizer(r'\\w+')\n  string = tokenizer.tokenize(string.lower())\n  list_tokens_elements = [element for element in string if len(element)>1]\n  return list_tokens_elements","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.125215Z","iopub.execute_input":"2022-08-04T08:27:11.125559Z","iopub.status.idle":"2022-08-04T08:27:11.132109Z","shell.execute_reply.started":"2022-08-04T08:27:11.125526Z","shell.execute_reply":"2022-08-04T08:27:11.130429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_effective_args['discourse_text_tokenized'] = df_effective_args['discourse_text'].apply(create_token)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.133604Z","iopub.execute_input":"2022-08-04T08:27:11.134240Z","iopub.status.idle":"2022-08-04T08:27:11.898178Z","shell.execute_reply.started":"2022-08-04T08:27:11.134205Z","shell.execute_reply":"2022-08-04T08:27:11.897331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_effective_args.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.899383Z","iopub.execute_input":"2022-08-04T08:27:11.899697Z","iopub.status.idle":"2022-08-04T08:27:11.912344Z","shell.execute_reply.started":"2022-08-04T08:27:11.899673Z","shell.execute_reply":"2022-08-04T08:27:11.911212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lemmatization","metadata":{}},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()\n\ndef create_lemmatization(list_of_words_tokenized) :\n  list_of_words_tokenized_lemmantized = []\n  for token in list_of_words_tokenized:\n    lemmetized_word = lemmatizer.lemmatize(token)\n    if len(lemmetized_word) > 1 : \n      list_of_words_tokenized_lemmantized.append(lemmetized_word)\n  return list_of_words_tokenized_lemmantized\n\ndf_effective_args['discourse_text_lemmatized'] = df_effective_args['discourse_text_tokenized'].apply(create_lemmatization)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:11.915615Z","iopub.execute_input":"2022-08-04T08:27:11.916015Z","iopub.status.idle":"2022-08-04T08:27:18.662658Z","shell.execute_reply.started":"2022-08-04T08:27:11.915988Z","shell.execute_reply":"2022-08-04T08:27:18.661816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Stopwords corpus + nltk","metadata":{}},{"cell_type":"code","source":"list_all_words = [j for i in df_effective_args['discourse_text_lemmatized'].tolist() for j in i]\ndictionnaire_des_frequences = nltk.FreqDist(list_all_words)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:18.663646Z","iopub.execute_input":"2022-08-04T08:27:18.663887Z","iopub.status.idle":"2022-08-04T08:27:19.680282Z","shell.execute_reply.started":"2022-08-04T08:27:18.663849Z","shell.execute_reply":"2022-08-04T08:27:19.679346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stop_words_corpus = []\nnombre_de_mot_selectionne = 25\nfor i in range(0, nombre_de_mot_selectionne):\n  stop_words_corpus.append(dictionnaire_des_frequences.most_common(nombre_de_mot_selectionne+1)[i][0])\nstop_words_nltk = list(set(stopwords.words('english')))\nstop_words_merged = list(set(stop_words_nltk+stop_words_corpus))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:19.682367Z","iopub.execute_input":"2022-08-04T08:27:19.682675Z","iopub.status.idle":"2022-08-04T08:27:19.827781Z","shell.execute_reply.started":"2022-08-04T08:27:19.682645Z","shell.execute_reply":"2022-08-04T08:27:19.827080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(stop_words_corpus)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:19.828667Z","iopub.execute_input":"2022-08-04T08:27:19.829369Z","iopub.status.idle":"2022-08-04T08:27:19.834448Z","shell.execute_reply.started":"2022-08-04T08:27:19.829343Z","shell.execute_reply":"2022-08-04T08:27:19.833281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## delete stop words corpus","metadata":{}},{"cell_type":"code","source":"def delete_stop_words_in_corpus(x, list_of_stop_words = stop_words_merged) :\n  y = x.copy()\n  for element in list_of_stop_words :\n    while element in y: y.remove(element)\n  return y","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:19.835522Z","iopub.execute_input":"2022-08-04T08:27:19.835836Z","iopub.status.idle":"2022-08-04T08:27:19.845171Z","shell.execute_reply.started":"2022-08-04T08:27:19.835810Z","shell.execute_reply":"2022-08-04T08:27:19.844209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_effective_args['no_stop_words'] = df_effective_args['discourse_text_lemmatized'].transform(delete_stop_words_in_corpus)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:19.847964Z","iopub.execute_input":"2022-08-04T08:27:19.848592Z","iopub.status.idle":"2022-08-04T08:27:24.464823Z","shell.execute_reply.started":"2022-08-04T08:27:19.848557Z","shell.execute_reply":"2022-08-04T08:27:24.463921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Delete some token to accelerate computing time","metadata":{}},{"cell_type":"code","source":"def delete_y_words(nb_words_to_save, liste_words):\n\n  dictionnaire_des_frequences = nltk.FreqDist(liste_words)\n  dictionnaire_des_frequences = dictionnaire_des_frequences.most_common(nb_words_to_save)\n  words = []\n  for i in range(0, nb_words_to_save):\n    words.append(dictionnaire_des_frequences[i][0])\n\n  return words\n\nlist_all_words = [j for i in df_effective_args['no_stop_words'].tolist() for j in i]\nnb_words_to_save = 2000\nliste_words_to_keep = delete_y_words(nb_words_to_save, list_all_words)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:24.465894Z","iopub.execute_input":"2022-08-04T08:27:24.466146Z","iopub.status.idle":"2022-08-04T08:27:25.016184Z","shell.execute_reply.started":"2022-08-04T08:27:24.466123Z","shell.execute_reply":"2022-08-04T08:27:25.014739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def delete_if_not_in(liste, liste_words_to_keep = liste_words_to_keep):\n\n  words_in_list = [i for i in liste if i in liste_words_to_keep]\n\n  return list(words_in_list)\n\ndf_effective_args['only_'+str(nb_words_to_save)+'_tokens'] = df_effective_args['no_stop_words'].apply(delete_if_not_in)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:25.019777Z","iopub.execute_input":"2022-08-04T08:27:25.020109Z","iopub.status.idle":"2022-08-04T08:27:31.955733Z","shell.execute_reply.started":"2022-08-04T08:27:25.020081Z","shell.execute_reply":"2022-08-04T08:27:31.954778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"def counter_len_in_text(string):\n  return len(string)\n\ndf_effective_args['Longueur_texte'] = df_effective_args['only_'+str(nb_words_to_save)+'_tokens'].apply(counter_len_in_text)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:31.957289Z","iopub.execute_input":"2022-08-04T08:27:31.957652Z","iopub.status.idle":"2022-08-04T08:27:31.977011Z","shell.execute_reply.started":"2022-08-04T08:27:31.957618Z","shell.execute_reply":"2022-08-04T08:27:31.976111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print((100-100*(df_effective_args.isna().sum()/(df_effective_args.shape[0]))).sort_values(ascending=False))\nprint(df_effective_args.dtypes.value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:31.978183Z","iopub.execute_input":"2022-08-04T08:27:31.978501Z","iopub.status.idle":"2022-08-04T08:27:32.004742Z","shell.execute_reply.started":"2022-08-04T08:27:31.978470Z","shell.execute_reply":"2022-08-04T08:27:32.003921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\ndf_effective_args['Longueur_texte'].hist(bins=150)\nplt.axvline(df_effective_args['Longueur_texte'].mean(), color='g', linestyle='dashed', linewidth=1)\nplt.axvline(df_effective_args['Longueur_texte'].median(), color='r', linestyle='dashed', linewidth=1)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:32.006391Z","iopub.execute_input":"2022-08-04T08:27:32.007045Z","iopub.status.idle":"2022-08-04T08:27:33.053049Z","shell.execute_reply.started":"2022-08-04T08:27:32.007011Z","shell.execute_reply":"2022-08-04T08:27:33.052133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of message have small number of words","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 4))\nplt.xlim(0, 150)\nsns.histplot(data=df_effective_args, x=df_effective_args['Longueur_texte'], hue=df_effective_args['discourse_effectiveness'], bins=100, multiple=\"stack\")\nplt.show()\n\nplt.figure(figsize=(8, 4))\nsns.displot(data=df_effective_args, x=df_effective_args['Longueur_texte'], hue=df_effective_args['discourse_effectiveness'], kind=\"kde\")\nplt.xlim(0, 150)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:33.054128Z","iopub.execute_input":"2022-08-04T08:27:33.054924Z","iopub.status.idle":"2022-08-04T08:27:34.285516Z","shell.execute_reply.started":"2022-08-04T08:27:33.054891Z","shell.execute_reply":"2022-08-04T08:27:34.284239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure()\nsns.boxplot(y=df_effective_args['Longueur_texte'], x=df_effective_args['discourse_effectiveness'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:27:34.287022Z","iopub.execute_input":"2022-08-04T08:27:34.288081Z","iopub.status.idle":"2022-08-04T08:27:34.423594Z","shell.execute_reply.started":"2022-08-04T08:27:34.288048Z","shell.execute_reply":"2022-08-04T08:27:34.422910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The longer the text, the more effective it is","metadata":{}},{"cell_type":"markdown","source":"# Feature engineering","metadata":{}},{"cell_type":"code","source":"df_effective_args['type_and_tokens'] = df_effective_args['discourse_type'].map(lambda i: [i]) + df_effective_args['only_'+str(nb_words_to_save)+'_tokens']","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:29:06.244175Z","iopub.execute_input":"2022-08-04T08:29:06.244522Z","iopub.status.idle":"2022-08-04T08:29:06.309878Z","shell.execute_reply.started":"2022-08-04T08:29:06.244495Z","shell.execute_reply":"2022-08-04T08:29:06.309115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TF IDF","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer, TfidfTransformer\nfrom sklearn.preprocessing import MultiLabelBinarizer, LabelBinarizer\nfrom sklearn.decomposition import TruncatedSVD, PCA\nfrom sklearn.preprocessing import StandardScaler, RobustScaler\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.model_selection import train_test_split, GridSearchCV, learning_curve\nfrom sklearn.svm import LinearSVC\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import r2_score, precision_score, recall_score, f1_score, classification_report, plot_roc_curve\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.multiclass import OneVsRestClassifier, OneVsOneClassifier\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, VotingClassifier\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.model_selection import train_test_split, GridSearchCV, learning_curve\n\nimport joblib\nimport ast\nimport time\n\nfrom sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:28:11.307681Z","iopub.execute_input":"2022-08-04T08:28:11.308056Z","iopub.status.idle":"2022-08-04T08:28:11.316761Z","shell.execute_reply.started":"2022-08-04T08:28:11.308027Z","shell.execute_reply":"2022-08-04T08:28:11.315569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparation train and test set ","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(df_effective_args['type_and_tokens'], df_effective_args['discourse_effectiveness'], test_size=0.2, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:32:56.825660Z","iopub.execute_input":"2022-08-04T08:32:56.826074Z","iopub.status.idle":"2022-08-04T08:32:56.839381Z","shell.execute_reply.started":"2022-08-04T08:32:56.826043Z","shell.execute_reply":"2022-08-04T08:32:56.838636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_words = X_train.tolist()\nlist_corpus = []\nfor element in list_words :\n  list_corpus.append(' '.join(element))\ntfidftvecto =  TfidfVectorizer()\ntfidf_model = tfidftvecto.fit(list_corpus)\ntfidf_tokens = tfidftvecto.get_feature_names_out()\nX_train_encoded = pd.DataFrame(data = tfidftvecto.fit_transform(list_corpus).toarray(),columns = tfidf_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:32:56.864268Z","iopub.execute_input":"2022-08-04T08:32:56.864939Z","iopub.status.idle":"2022-08-04T08:32:58.313574Z","shell.execute_reply.started":"2022-08-04T08:32:56.864902Z","shell.execute_reply":"2022-08-04T08:32:58.312223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_words_test = X_test.tolist()\nlist_corpus_test = []\nfor element in list_words_test :\n  list_corpus_test.append(' '.join(element))\nX_test_encoded = pd.DataFrame(data = tfidf_model.transform(list_corpus_test).toarray(),columns = tfidf_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:32:58.315213Z","iopub.execute_input":"2022-08-04T08:32:58.315522Z","iopub.status.idle":"2022-08-04T08:32:58.622615Z","shell.execute_reply.started":"2022-08-04T08:32:58.315494Z","shell.execute_reply":"2022-08-04T08:32:58.621666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lb = LabelBinarizer()\ndf_labels_train = pd.DataFrame(lb.fit_transform(y_train), columns=lb.classes_)\ndf_labels_test = pd.DataFrame(lb.fit_transform(y_test), columns=lb.classes_)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:32:58.623765Z","iopub.execute_input":"2022-08-04T08:32:58.624066Z","iopub.status.idle":"2022-08-04T08:32:58.703056Z","shell.execute_reply.started":"2022-08-04T08:32:58.624040Z","shell.execute_reply":"2022-08-04T08:32:58.701838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numerical_features = list(X_train_encoded.select_dtypes(['int']).columns)+list(X_train_encoded.select_dtypes(['float']).columns)\nnumerical_pipeline = make_pipeline(StandardScaler())\n\nL_R = LogisticRegression(max_iter= 1000)\nmultilabel_classifier = OneVsRestClassifier(L_R)\npreprocessor = make_column_transformer((numerical_pipeline, numerical_features))\nmodel_L_R = make_pipeline(preprocessor, multilabel_classifier)\n\nstart_time_fit = time.time()\nmodel_L_R.fit(X_train_encoded, df_labels_train)\n\nelapsed_time_fit = time.time() - start_time_fit #time\n\nstart_time_predict = time.time()\ndf_labels_pred = model_L_R.predict(X_test_encoded)\n\nelapsed_time_predict = time.time() - start_time_predict #time\nprint(elapsed_time_fit, elapsed_time_predict) #time\n\nprint(classification_report(df_labels_test, df_labels_pred))","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:32:58.705086Z","iopub.execute_input":"2022-08-04T08:32:58.705379Z","iopub.status.idle":"2022-08-04T08:33:14.724419Z","shell.execute_reply.started":"2022-08-04T08:32:58.705353Z","shell.execute_reply":"2022-08-04T08:33:14.720749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission ","metadata":{}},{"cell_type":"code","source":"def pipeline_from_dataset_to_tokens(df, nb_words_to_save):\n  df['discourse_text_tokenized'] = df['discourse_text'].apply(create_token)\n  df['discourse_text_lemmatized'] = df['discourse_text_tokenized'].apply(create_lemmatization)\n  df['no_stop_words'] = df['discourse_text_lemmatized'].transform(delete_stop_words_in_corpus)\n  df['only_'+str(nb_words_to_save)+'_tokens'] = df['no_stop_words'].apply(delete_if_not_in)\n  df['Longueur_texte'] = df['only_'+str(nb_words_to_save)+'_tokens'].apply(counter_len_in_text)\n  df['type_and_tokens'] = df['discourse_type'].map(lambda i: [i]) + df['only_'+str(nb_words_to_save)+'_tokens']\n  #df = df[df['Longueur_texte']<150]\n  print(df.shape)\n  return df","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:33:14.725826Z","iopub.execute_input":"2022-08-04T08:33:14.726290Z","iopub.status.idle":"2022-08-04T08:33:14.735159Z","shell.execute_reply.started":"2022-08-04T08:33:14.726259Z","shell.execute_reply":"2022-08-04T08:33:14.733147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_needed_to_predict = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ndf_needed_to_predict_take_token = pipeline_from_dataset_to_tokens(df_needed_to_predict, nb_words_to_save = nb_words_to_save)\nX_to_predict = df_needed_to_predict_take_token['type_and_tokens']","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:33:14.736126Z","iopub.execute_input":"2022-08-04T08:33:14.736465Z","iopub.status.idle":"2022-08-04T08:33:14.770448Z","shell.execute_reply.started":"2022-08-04T08:33:14.736431Z","shell.execute_reply":"2022-08-04T08:33:14.769690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_words_test = X_to_predict.tolist()\nlist_corpus_test = []\nfor element in list_words_test :\n  list_corpus_test.append(' '.join(element))\nX_to_predict_encoded = pd.DataFrame(data = tfidf_model.transform(list_corpus_test).toarray(),columns = tfidf_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:33:14.774335Z","iopub.execute_input":"2022-08-04T08:33:14.776225Z","iopub.status.idle":"2022-08-04T08:33:14.784375Z","shell.execute_reply.started":"2022-08-04T08:33:14.776194Z","shell.execute_reply":"2022-08-04T08:33:14.783653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_predicted = pd.DataFrame(model_L_R.predict_proba(X_to_predict_encoded), columns = list(lb.classes_))\ndf_predicted['discourse_id'] = df_needed_to_predict['discourse_id']\ndf_predicted = df_predicted[['discourse_id' ,'Ineffective', 'Adequate', 'Effective']]\ndf_predicted.to_csv('submission.csv',index=False)\ndf_predicted","metadata":{"execution":{"iopub.status.busy":"2022-08-04T08:33:14.787726Z","iopub.execute_input":"2022-08-04T08:33:14.789709Z","iopub.status.idle":"2022-08-04T08:33:14.833853Z","shell.execute_reply.started":"2022-08-04T08:33:14.789675Z","shell.execute_reply":"2022-08-04T08:33:14.833172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# switch of internet ","metadata":{}},{"cell_type":"markdown","source":"To be able to submit you have to switch of internet connection \n*   Settings\n*   Internet off (-)","metadata":{}},{"cell_type":"markdown","source":"# In case of submission errors ","metadata":{}},{"cell_type":"markdown","source":"Advice extracted from another topic : replace test.csv by train.csv. This trick could be helpful to spot errors.","metadata":{}},{"cell_type":"markdown","source":"# Documentation","metadata":{}},{"cell_type":"markdown","source":"In this following notebook : \nhttps://www.kaggle.com/code/iamleonie/feedback-prize-eda-starter-for-beginners\nyou will find interesting cleaning (presence of duplicates) and very comlete EDA.","metadata":{}},{"cell_type":"markdown","source":"In this one : Tips and Tricks from past text classification competitions\nhttps://www.kaggle.com/competitions/feedback-prize-effectiveness/discussion/335896","metadata":{}}]}