{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Reading & Exploring Data:","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:52.371546Z","iopub.execute_input":"2023-09-01T04:16:52.371948Z","iopub.status.idle":"2023-09-01T04:16:52.377367Z","shell.execute_reply.started":"2023-09-01T04:16:52.371918Z","shell.execute_reply":"2023-09-01T04:16:52.376012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/quora-insincere-questions-classification'\nos.listdir(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:52.384303Z","iopub.execute_input":"2023-09-01T04:16:52.385414Z","iopub.status.idle":"2023-09-01T04:16:52.406923Z","shell.execute_reply.started":"2023-09-01T04:16:52.385367Z","shell.execute_reply":"2023-09-01T04:16:52.405970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_fname = os.path.join(file_path, 'train.csv')\ntest_fname = os.path.join(file_path, 'test.csv')\nsubmission_fname = os.path.join(file_path, 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:52.408832Z","iopub.execute_input":"2023-09-01T04:16:52.410209Z","iopub.status.idle":"2023-09-01T04:16:52.416496Z","shell.execute_reply.started":"2023-09-01T04:16:52.410163Z","shell.execute_reply":"2023-09-01T04:16:52.415503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df = pd.read_csv(train_fname)\nraw_test_df = pd.read_csv(test_fname)\nsubmission_df = pd.read_csv(submission_fname)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:52.418005Z","iopub.execute_input":"2023-09-01T04:16:52.418378Z","iopub.status.idle":"2023-09-01T04:16:57.350668Z","shell.execute_reply.started":"2023-09-01T04:16:52.418348Z","shell.execute_reply":"2023-09-01T04:16:57.349485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.352300Z","iopub.execute_input":"2023-09-01T04:16:57.352641Z","iopub.status.idle":"2023-09-01T04:16:57.365814Z","shell.execute_reply.started":"2023-09-01T04:16:57.352613Z","shell.execute_reply":"2023-09-01T04:16:57.364583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['target'].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.370258Z","iopub.execute_input":"2023-09-01T04:16:57.370684Z","iopub.status.idle":"2023-09-01T04:16:57.396161Z","shell.execute_reply.started":"2023-09-01T04:16:57.370626Z","shell.execute_reply":"2023-09-01T04:16:57.394899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['target'].value_counts(normalize=True).plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.397867Z","iopub.execute_input":"2023-09-01T04:16:57.398353Z","iopub.status.idle":"2023-09-01T04:16:57.656962Z","shell.execute_reply.started":"2023-09-01T04:16:57.398311Z","shell.execute_reply":"2023-09-01T04:16:57.656042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_df = raw_train_df[raw_train_df['target']==0].copy()\ninsincere_df = raw_train_df[raw_train_df['target']==1].copy()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.658386Z","iopub.execute_input":"2023-09-01T04:16:57.659458Z","iopub.status.idle":"2023-09-01T04:16:57.846755Z","shell.execute_reply.started":"2023-09-01T04:16:57.659423Z","shell.execute_reply":"2023-09-01T04:16:57.845559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.848248Z","iopub.execute_input":"2023-09-01T04:16:57.849164Z","iopub.status.idle":"2023-09-01T04:16:57.858623Z","shell.execute_reply.started":"2023-09-01T04:16:57.849110Z","shell.execute_reply":"2023-09-01T04:16:57.857177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.860130Z","iopub.execute_input":"2023-09-01T04:16:57.860561Z","iopub.status.idle":"2023-09-01T04:16:57.874263Z","shell.execute_reply.started":"2023-09-01T04:16:57.860531Z","shell.execute_reply":"2023-09-01T04:16:57.871783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_test_df","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.877040Z","iopub.execute_input":"2023-09-01T04:16:57.877523Z","iopub.status.idle":"2023-09-01T04:16:57.897475Z","shell.execute_reply.started":"2023-09-01T04:16:57.877479Z","shell.execute_reply":"2023-09-01T04:16:57.896094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.899238Z","iopub.execute_input":"2023-09-01T04:16:57.900322Z","iopub.status.idle":"2023-09-01T04:16:57.924091Z","shell.execute_reply.started":"2023-09-01T04:16:57.900284Z","shell.execute_reply":"2023-09-01T04:16:57.922127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Preprocessing Technique:","metadata":{}},{"cell_type":"markdown","source":"### TF-IDF (Term Frequency-Inverse Document Frequency):\n<p>\n    \nTF-IDF is a feature extraction technique for textual data. It not only converts documents or text into vectors of numeric values by means of count but aims to capture the importance of words in a collection of documents, so that we can feed the meaningful representation of text into Machine Learning model. Machine Learning models cannot process text directly that's why we first need to convert to have fixed length input for modeling. Here's a step-by-step explanation:\n\n<ol>\n    <li><strong>Term Frequency (TF):</strong><br>\n        Term Frequency (TF) provides information about the local importance of a word within a single document by measuring how often a word appears in a specific document. It is calculated using the formula:\n    </li>\n    <br>\n    <p>TF(word, document) = (Number of times the word appears in the document) / (Total number of words in the document)</p>\n    <br>\n    <li><strong>Inverse Document Frequency (IDF):</strong><br>\n        Inverse Document Frequency (IDF) quantifies how unique or important a word is across the entire corpus of documents by emphasizing words that appear in fewer documents, giving higher weights to rare and distinctive words. It's calculated using the formula:\n    </li>\n    <br>\n    <p>IDF(word) = log((Total number of documents) / (Number of documents containing the word))</p>\n    <br>\n    <li><strong>TF-IDF Calculation:</strong><br>\n        The TF-IDF score for a word in a document is the product of its Term Frequency (TF) and Inverse Document Frequency (IDF). The resulting value reflects how important a word is in a particular document relative to its occurrence across the entire corpus.\n    </li>\n    <br>\n    <p>TF-IDF(word, document) = TF(word, document) * IDF(word)</p>\n</ol>\n<br>\n\n\n\n<strong>Comparison with Bag of Words (BoW):</strong>\n\n<ul>\n    <li>Handling of Word Importance: BoW only considers the presence or absence of words and their counts in documents.TF-IDF takes into account both the frequency of a word within a document (TF) and its rarity across the entire corpus (IDF), providing a more nuanced representation of word importance.\n    </li>\n    <br>\n    <li>Impact of Common Words: BoW representation can give high weights to common words (e.g., \"the,\" \"and\") that appear frequently across documents. TF-IDF reduces the weight of common words by factoring in their inverse document frequency.\n    </li>\n    <br>\n    <li>Handling of Rare Words: BoW doesn't distinguish rare words from common ones, treating all words equally. TF-IDF gives higher weights to rare words, as their IDF values are higher.\n    </li>\n    <br>\n    <li>Sparsity: BoW vectors represent documents as counts of individual words. Since most documents won't contain all the words in the vocabulary, BoW vectors are often sparse due to the many zero-count entries. TF-IDF vectors attempt to weigh words based on their importance in a document relative to the entire corpus. However, if a word is not present in a document (TF = 0), its TF-IDF score for that document will indeed be 0, regardless of its IDF. This can result in sparse TF-IDF vectors, similar to BoW vectors.\n    </li>\n    <br>\n    <li>Contextual Information: BoW completely discards word order and context, treating documents as bags of words. TF-IDF retains some context information by considering both the local (TF) and global (IDF) significance of words.\n</ul>\n    \n    \nIn summary, while both BoW and TF-IDF are methods for converting text into numerical representations, TF-IDF provides a more refined approach by considering the importance of words in both local and global contexts. It addresses some of the limitations of BoW and offers more meaningful representations, especially when dealing with larger corpora and documents.\n\n\n\n</p>","metadata":{}},{"cell_type":"markdown","source":"### Configure Count Vectorizer Parameters:","metadata":{}},{"cell_type":"code","source":"import nltk","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:57.926061Z","iopub.execute_input":"2023-09-01T04:16:57.926903Z","iopub.status.idle":"2023-09-01T04:16:59.351311Z","shell.execute_reply.started":"2023-09-01T04:16:57.926867Z","shell.execute_reply":"2023-09-01T04:16:59.350148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# word tokenizer\nfrom nltk import word_tokenize\nnltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.352613Z","iopub.execute_input":"2023-09-01T04:16:59.353038Z","iopub.status.idle":"2023-09-01T04:16:59.692257Z","shell.execute_reply.started":"2023-09-01T04:16:59.352994Z","shell.execute_reply":"2023-09-01T04:16:59.690733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting list of english stopwords\nfrom nltk.corpus import stopwords\n\nnltk.download('stopwords')\nen_stopwords = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.699261Z","iopub.execute_input":"2023-09-01T04:16:59.699650Z","iopub.status.idle":"2023-09-01T04:16:59.718034Z","shell.execute_reply.started":"2023-09-01T04:16:59.699621Z","shell.execute_reply":"2023-09-01T04:16:59.715730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating stemmer object\nfrom nltk.stem import SnowballStemmer\nstemmer = SnowballStemmer(language=\"english\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.720047Z","iopub.execute_input":"2023-09-01T04:16:59.720556Z","iopub.status.idle":"2023-09-01T04:16:59.726577Z","shell.execute_reply.started":"2023-09-01T04:16:59.720512Z","shell.execute_reply":"2023-09-01T04:16:59.725202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.727731Z","iopub.execute_input":"2023-09-01T04:16:59.728082Z","iopub.status.idle":"2023-09-01T04:16:59.741882Z","shell.execute_reply.started":"2023-09-01T04:16:59.728053Z","shell.execute_reply":"2023-09-01T04:16:59.740415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize_and_stem(text):\n    return [stemmer.stem(token) for token in word_tokenize(text)]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.744424Z","iopub.execute_input":"2023-09-01T04:16:59.744786Z","iopub.status.idle":"2023-09-01T04:16:59.760012Z","shell.execute_reply.started":"2023-09-01T04:16:59.744757Z","shell.execute_reply":"2023-09-01T04:16:59.758880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating count vectorizer object with configuration to incorporate preprocessing techniques together\nvectorizer = TfidfVectorizer(lowercase=True,\n                             stop_words=en_stopwords, #removing stopwords\n                             tokenizer=tokenize_and_stem #tokenization and stemming at the same time\n                             )","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:16:59.763265Z","iopub.execute_input":"2023-09-01T04:16:59.763779Z","iopub.status.idle":"2023-09-01T04:16:59.775874Z","shell.execute_reply.started":"2023-09-01T04:16:59.763737Z","shell.execute_reply":"2023-09-01T04:16:59.774698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvectorizer.fit(raw_train_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:20:13.807519Z","iopub.execute_input":"2023-09-01T04:20:13.807918Z","iopub.status.idle":"2023-09-01T04:30:53.300034Z","shell.execute_reply.started":"2023-09-01T04:20:13.807887Z","shell.execute_reply":"2023-09-01T04:30:53.298650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vectorizer.vocabulary_)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:30:53.302708Z","iopub.execute_input":"2023-09-01T04:30:53.303249Z","iopub.status.idle":"2023-09-01T04:30:53.311226Z","shell.execute_reply.started":"2023-09-01T04:30:53.303205Z","shell.execute_reply":"2023-09-01T04:30:53.309948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:30:53.312846Z","iopub.execute_input":"2023-09-01T04:30:53.313213Z","iopub.status.idle":"2023-09-01T04:30:53.605097Z","shell.execute_reply.started":"2023-09-01T04:30:53.313185Z","shell.execute_reply":"2023-09-01T04:30:53.603909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel_inputs = vectorizer.transform(raw_train_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:30:53.607768Z","iopub.execute_input":"2023-09-01T04:30:53.608175Z","iopub.status.idle":"2023-09-01T04:41:37.093301Z","shell.execute_reply.started":"2023-09-01T04:30:53.608118Z","shell.execute_reply":"2023-09-01T04:41:37.091762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:41:37.095184Z","iopub.execute_input":"2023-09-01T04:41:37.095760Z","iopub.status.idle":"2023-09-01T04:41:37.103253Z","shell.execute_reply.started":"2023-09-01T04:41:37.095727Z","shell.execute_reply":"2023-09-01T04:41:37.101688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:41:37.105550Z","iopub.execute_input":"2023-09-01T04:41:37.105951Z","iopub.status.idle":"2023-09-01T04:41:37.118978Z","shell.execute_reply.started":"2023-09-01T04:41:37.105919Z","shell.execute_reply":"2023-09-01T04:41:37.118099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['question_text'].values[0]","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:41:37.120078Z","iopub.execute_input":"2023-09-01T04:41:37.120929Z","iopub.status.idle":"2023-09-01T04:41:37.134423Z","shell.execute_reply.started":"2023-09-01T04:41:37.120895Z","shell.execute_reply":"2023-09-01T04:41:37.132913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs[0].toarray()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:41:37.135901Z","iopub.execute_input":"2023-09-01T04:41:37.136312Z","iopub.status.idle":"2023-09-01T04:41:37.152849Z","shell.execute_reply.started":"2023-09-01T04:41:37.136281Z","shell.execute_reply":"2023-09-01T04:41:37.151703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_inputs = vectorizer.transform(raw_test_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:41:37.155251Z","iopub.execute_input":"2023-09-01T04:41:37.155757Z","iopub.status.idle":"2023-09-01T04:44:40.065490Z","shell.execute_reply.started":"2023-09-01T04:41:37.155708Z","shell.execute_reply":"2023-09-01T04:44:40.064246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training:","metadata":{}},{"cell_type":"code","source":"# Splitting data into train_set & validation_set\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:21.820829Z","iopub.execute_input":"2023-09-01T04:45:21.821372Z","iopub.status.idle":"2023-09-01T04:45:21.826910Z","shell.execute_reply.started":"2023-09-01T04:45:21.821329Z","shell.execute_reply":"2023-09-01T04:45:21.825707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs, valid_inputs, train_target, valid_target = train_test_split(model_inputs, raw_train_df['target'], test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:22.193543Z","iopub.execute_input":"2023-09-01T04:45:22.193963Z","iopub.status.idle":"2023-09-01T04:45:22.563445Z","shell.execute_reply.started":"2023-09-01T04:45:22.193931Z","shell.execute_reply":"2023-09-01T04:45:22.562194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs.shape, valid_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:22.593104Z","iopub.execute_input":"2023-09-01T04:45:22.593759Z","iopub.status.idle":"2023-09-01T04:45:22.600218Z","shell.execute_reply.started":"2023-09-01T04:45:22.593724Z","shell.execute_reply":"2023-09-01T04:45:22.598930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_target.shape, valid_target.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:22.781563Z","iopub.execute_input":"2023-09-01T04:45:22.782024Z","iopub.status.idle":"2023-09-01T04:45:22.789924Z","shell.execute_reply.started":"2023-09-01T04:45:22.781991Z","shell.execute_reply":"2023-09-01T04:45:22.788645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_target.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:22.953451Z","iopub.execute_input":"2023-09-01T04:45:22.953875Z","iopub.status.idle":"2023-09-01T04:45:22.972971Z","shell.execute_reply.started":"2023-09-01T04:45:22.953842Z","shell.execute_reply":"2023-09-01T04:45:22.971679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_target.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:24.371295Z","iopub.execute_input":"2023-09-01T04:45:24.372079Z","iopub.status.idle":"2023-09-01T04:45:24.388489Z","shell.execute_reply.started":"2023-09-01T04:45:24.372035Z","shell.execute_reply":"2023-09-01T04:45:24.386189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression Model:","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:25.684293Z","iopub.execute_input":"2023-09-01T04:45:25.684725Z","iopub.status.idle":"2023-09-01T04:45:25.690164Z","shell.execute_reply.started":"2023-09-01T04:45:25.684694Z","shell.execute_reply":"2023-09-01T04:45:25.689241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_ITER = 1000","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:25.933511Z","iopub.execute_input":"2023-09-01T04:45:25.934652Z","iopub.status.idle":"2023-09-01T04:45:25.939431Z","shell.execute_reply.started":"2023-09-01T04:45:25.934603Z","shell.execute_reply":"2023-09-01T04:45:25.938602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_model = LogisticRegression(max_iter=MAX_ITER)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:26.264045Z","iopub.execute_input":"2023-09-01T04:45:26.264898Z","iopub.status.idle":"2023-09-01T04:45:26.271251Z","shell.execute_reply.started":"2023-09-01T04:45:26.264853Z","shell.execute_reply":"2023-09-01T04:45:26.269577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlog_reg_model.fit(train_inputs, train_target)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:45:26.815080Z","iopub.execute_input":"2023-09-01T04:45:26.815500Z","iopub.status.idle":"2023-09-01T04:46:37.070958Z","shell.execute_reply.started":"2023-09-01T04:45:26.815469Z","shell.execute_reply":"2023-09-01T04:46:37.069000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_preds = log_reg_model.predict(valid_inputs)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:37.080189Z","iopub.execute_input":"2023-09-01T04:46:37.080875Z","iopub.status.idle":"2023-09-01T04:46:37.127454Z","shell.execute_reply.started":"2023-09-01T04:46:37.080823Z","shell.execute_reply":"2023-09-01T04:46:37.126354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:37.131865Z","iopub.execute_input":"2023-09-01T04:46:37.132224Z","iopub.status.idle":"2023-09-01T04:46:37.137017Z","shell.execute_reply.started":"2023-09-01T04:46:37.132195Z","shell.execute_reply":"2023-09-01T04:46:37.135811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_accuracy = accuracy_score(valid_target, valid_preds)\nvalid_precision = precision_score(valid_target, valid_preds)\nvalid_recall = recall_score(valid_target, valid_preds)\nvalid_f1 = f1_score(valid_target, valid_preds)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:37.139725Z","iopub.execute_input":"2023-09-01T04:46:37.140467Z","iopub.status.idle":"2023-09-01T04:46:37.614127Z","shell.execute_reply.started":"2023-09-01T04:46:37.140424Z","shell.execute_reply":"2023-09-01T04:46:37.612828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_accuracy, valid_precision, valid_recall, valid_f1","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:37.615553Z","iopub.execute_input":"2023-09-01T04:46:37.615882Z","iopub.status.idle":"2023-09-01T04:46:37.623798Z","shell.execute_reply.started":"2023-09-01T04:46:37.615854Z","shell.execute_reply":"2023-09-01T04:46:37.622552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:47.479421Z","iopub.execute_input":"2023-09-01T04:46:47.479836Z","iopub.status.idle":"2023-09-01T04:46:47.485385Z","shell.execute_reply.started":"2023-09-01T04:46:47.479803Z","shell.execute_reply":"2023-09-01T04:46:47.483980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_preds = np.random.choice((0, 1), len(valid_target))","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:47.775384Z","iopub.execute_input":"2023-09-01T04:46:47.775786Z","iopub.status.idle":"2023-09-01T04:46:47.782369Z","shell.execute_reply.started":"2023-09-01T04:46:47.775746Z","shell.execute_reply":"2023-09-01T04:46:47.781396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_accuracy = accuracy_score(valid_target, random_preds)\nrandom_f1 = f1_score(valid_target, random_preds)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:48.109353Z","iopub.execute_input":"2023-09-01T04:46:48.109732Z","iopub.status.idle":"2023-09-01T04:46:48.331419Z","shell.execute_reply.started":"2023-09-01T04:46:48.109704Z","shell.execute_reply":"2023-09-01T04:46:48.330199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_accuracy, random_f1","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:54.155389Z","iopub.execute_input":"2023-09-01T04:46:54.155765Z","iopub.status.idle":"2023-09-01T04:46:54.164000Z","shell.execute_reply.started":"2023-09-01T04:46:54.155736Z","shell.execute_reply":"2023-09-01T04:46:54.162735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission to Kaggle:","metadata":{}},{"cell_type":"code","source":"test_preds = log_reg_model.predict(test_inputs)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:58.233386Z","iopub.execute_input":"2023-09-01T04:46:58.233784Z","iopub.status.idle":"2023-09-01T04:46:58.259548Z","shell.execute_reply.started":"2023-09-01T04:46:58.233745Z","shell.execute_reply":"2023-09-01T04:46:58.258385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['prediction'] = test_preds","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:58.385264Z","iopub.execute_input":"2023-09-01T04:46:58.386015Z","iopub.status.idle":"2023-09-01T04:46:58.391221Z","shell.execute_reply.started":"2023-09-01T04:46:58.385980Z","shell.execute_reply":"2023-09-01T04:46:58.390236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['prediction'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:46:59.147797Z","iopub.execute_input":"2023-09-01T04:46:59.148706Z","iopub.status.idle":"2023-09-01T04:46:59.161221Z","shell.execute_reply.started":"2023-09-01T04:46:59.148657Z","shell.execute_reply":"2023-09-01T04:46:59.160031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T04:47:01.318219Z","iopub.execute_input":"2023-09-01T04:47:01.318633Z","iopub.status.idle":"2023-09-01T04:47:02.470533Z","shell.execute_reply.started":"2023-09-01T04:47:01.318601Z","shell.execute_reply":"2023-09-01T04:47:02.469463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}