{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Quora Text Classification\n\nOutline:\n- Download and explore data.\n- Apply text preprocessing techniques.\n- Implement the Bag of Words Model.\n- Train ML models for text classification.\n- Make predictions and submit to Kaggle.","metadata":{"id":"r92bwuNZrWwd"}},{"cell_type":"markdown","source":"## Download and Explore Data.","metadata":{"id":"F7puqejnsJPj"}},{"cell_type":"markdown","source":"Outline:\n\n1. Download data from kaggle to colab.\n2. Explore the data using pandas.\n3. creating a small working sample.","metadata":{"id":"FVJ1lFiSsuAb"}},{"cell_type":"markdown","source":"###### 1. Download the data to colab from kaggle.","metadata":{"id":"QlKKxANAtUiE"}},{"cell_type":"code","source":"!ls","metadata":{"id":"jn2D7Kfpsszy","outputId":"a38838fa-186b-4ada-87f8-aeb74e5c7e19"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"id":"JDLCv3Y-uk4L"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IS_KAGGLE = 'KAGGLE_KERNEL_RUN_TYPE' in os.environ","metadata":{"id":"TizqME1QmtN3"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if IS_KAGGLE:\n  data_dir = '../input/quora-insincere-questions-classification'\n  train_fname = data_dir + '/train.csv'\n  test_fname = data_dir + '/test.csv'\n  sample_fname = data_dir + '/sample_submission.csv'\n\nelse:\n  os.environ['KAGGLE_CONFIG_DIR'] = '.'\n  !kaggle competitions download -c quora-insincere-questions-classification -f train.csv -p data\n  !kaggle competitions download -c quora-insincere-questions-classification -f test.csv -p data\n  !kaggle competitions download -c quora-insincere-questions-classification -f sample_submission.csv -p data\n  train_fname = 'data/train.csv.zip'\n  test_fname = 'data/test.csv.zip'\n  sample_fname = 'data/sample_submission.csv.zip'\n\n# setting up environment variable which will tell the kaggle that where the config file is located\n# '.'  referrs to current dir\n# im telling kaggle that kaggle.json is located at current dir\n# ----------------------------------------------------------------\n# There are bunch of files in the dataset, so i want the specific file ie train.csv in data folder\n# now the data folder will be created\n#-------------------------------------------------------------------\n# as we can see these are zip files, you might be wondering about we need to unzip them\n# but pandas can directly work with the zipped files.","metadata":{"id":"XIHopJeSluJq","outputId":"fc7585b1-05da-4a9f-f0de-e09a63511c39"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### 2. Explore Data using pandas","metadata":{"id":"YwqzA0olwwtg"}},{"cell_type":"code","source":"import pandas as pd\n\nraw_df = pd.read_csv(train_fname)\nraw_df","metadata":{"id":"dILoqMcbwc-w","outputId":"ee6348c1-3898-41f4-9e02-cc92d7a365f2"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have question Id , Question text and target\n# target is 0 or 1.\n#    0 is sincere questions.\n#    1 are insincere questions.\n\n# lets create a sincere df first containing sincere\n\nsincere_df = raw_df[raw_df.target == 0]\n\n# lets first 10 que check sincere_df \n\nsincere_df.question_text.values[:10]","metadata":{"id":"3tYhVfTIygT8","outputId":"951e2a65-eed7-41e8-c2e8-ae8984daf0a6"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets create df for insincere questions\n\ninsincere_df = raw_df[raw_df.target == 1]\n\n# lets check 1st 10 insincere questions\n\ninsincere_df.question_text.values[:10]","metadata":{"id":"tUOnPjKNygYe","outputId":"343e012d-0f33-4877-db79-b9e6c0e81b3d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the sincere ques and insincere questions count using countplot\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\nprint(raw_df.target.value_counts())\nraw_df.target.value_counts(normalize=True).plot(kind = 'bar')\nplt.title('sincere questions vs insincere questions asked on Quara')\nplt.gcf().set_size_inches(10,5)","metadata":{"id":"iM8OFvuxyga8","outputId":"dcadd930-fc0e-4bd2-840e-396563af9e57"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(raw_df.target.value_counts(normalize=True)*100)","metadata":{"id":"6bpK7aUbygeX","outputId":"f2516525-7e11-4bc8-d71a-6b8e745a7fdf"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Almost 94 % questions asked are sincere.\n# and only 6% of the questions are insincere.\n\n# when evaluating a text based model accuracy is not the best metric.\n# as you may get 94 % based on getting 0.\n# so the competition asks for f1score","metadata":{"id":"Glq3PtgJ21Pc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(test_fname)\ntest_df","metadata":{"id":"ANjLgJuB4Ou6","outputId":"2ef7814f-43ae-4310-aa6a-015ccb55eb08"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Here the dataset is missing the target column \n# target needs to be predicted here, based on the training the model.","metadata":{"id":"5aZJQ0mp5xLz"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(sample_fname)\nsubmission_df","metadata":{"id":"-ARpalzH5xOh","outputId":"4ffb834d-9084-4dae-b5c8-4bea19098279"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###### 3. Creating a working sample","metadata":{"id":"pXfjMPGk9jaP"}},{"cell_type":"code","source":"sample_size = 100_000","metadata":{"id":"5g-e4w_093Ih"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# working with thousands of records is difficult.\n# so creating a sample df\n\nsample_df = raw_df.sample(sample_size, random_state = 50)\nsample_df","metadata":{"id":"e5BpmS1N5xRS","outputId":"2435aa65-0b85-4e38-b423-2ee2762802da"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so that was part 1.","metadata":{"id":"Mknka2nW5xU1"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Applying Text Preprocessing Techniques.\n\nOutline:\n\n1. Understand the Bag of Words model\n2. Tokenization\n3. Stop words\n4. Stemming\n","metadata":{"id":"-4QKLLJZsJuN"}},{"cell_type":"markdown","source":"##### 1. Understanding the Bag of Words model\n\n###### Bag of Intuition\n\n1. create a list of all the words across all the text documents.\n2. you convert each document into a vector count of each word.\n\n\nlimitations:\n1. there are too many words in a dataset.\n2. some words may occur too frequently\n3. some words may occur very rarely or only once\n4. a single word may have many forms (eg go, going, gone or bird vs birds)\n\n   so we may want to count the bird, birds, go, going, etc under the same root word and that is why we perform several word preprocessing techniques before creating the bag of word model\n\n   - first preprocessing technique is Tokenization:\n   \n   tokenization is splitting the document into words and separators(,.[])","metadata":{"id":"pBcT3_j-_Vy_"}},{"cell_type":"code","source":"# lets say we have 3 samples\n\n# 1. About the bird, the bird, bird bird bird\n# 2. You heard about the bird\n# 3. The bird is the word\n\n\n# now lets se the unique word in the dataset\n# these are the unique words:\n#       about  bird   heard   is    the    word   you\n#\n#\n# now we have to convert these text into a vector\n# because ML algo only work with nums\n# so how we convert the words into numeric vectors\n#\n# heres how:\n#\n# in sentence: About the bird, the bird, bird bird bird\n# how many times a word occurs\n#\n#   about  bird   heard   is    the    word   you\n#     1     5       0      0     2       0     0\n#\n# similarly for remaining sentences\n#   You heard about the bird\n#   about  bird   heard   is    the    word   you\n#     1     1       1      0     1       0     1\n#\n#   The bird is the word\n#   about  bird   heard   is    the    word   you\n#     0     1       0      1     2       1     0\n#\n# Now we are left with these vectors:\n#\n#   1     5       0      0     2       0     0\n#   1     1       1      0     1       0     1\n#   0     1       0      1     2       1     0\n#\n# these are called document vectors\n#\n# so that is how you build a bag of words model","metadata":{"id":"0dSjGyhv_U6c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets pick a question from a sincere_df\nq1 = sincere_df.question_text.values[1]\nq1","metadata":{"id":"_JdkoUVDsGsH","outputId":"36f6024d-f11b-4bf0-95a8-a5e6593875d5"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets pick an insincere question as well\nq2 = insincere_df.question_text.values[1]\nq2","metadata":{"id":"rt1b6ehGcj20","outputId":"6b13d231-205d-4156-99fb-2895ef3af91e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" import nltk\n from nltk.tokenize import word_tokenize\n nltk.download('punkt')","metadata":{"id":"jLKeXJZ3sGvQ","outputId":"d52c123b-6f59-4b92-bda9-e899f696068a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1","metadata":{"id":"17tLxtYKsGyn","outputId":"449ece34-2978-49cd-fa96-9cb891531422"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"word_tokenize(q1)","metadata":{"id":"P0jIwv1KhAdi","outputId":"7fa824a9-4e58-4203-d4e4-802f01c346b5"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# you can try with your own eg\n\nword_tokenize('what do you call me when your high?')","metadata":{"id":"lWNAlpXNhd23","outputId":"6511b74a-03ff-479d-f6ad-ff612251e426"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets tokenize q2 to see if theres any interesting tokenization\nword_tokenize(q2)","metadata":{"id":"Qrv47W2vh5gE","outputId":"3ee0e129-f375-4659-bd68-6a82e4a65612"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we can see that it separated each word","metadata":{"id":"HqwlrhzahaEA"}},{"cell_type":"code","source":"# lets store these tokenization \nq1_tokenize = word_tokenize(q1)\nq2_tokenize = word_tokenize(q2)","metadata":{"id":"ZNMseBXMiRDu"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Stop word removal\n\nnow what is stop word removal??\n\nit is removing commonly occuring words","metadata":{"id":"8FKTDQTLikdg"}},{"cell_type":"code","source":"q1_tokenize","metadata":{"id":"ewWFKMLZigu1","outputId":"594d2b73-0808-4a90-c6c9-06740bc6b8b8"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we use nltk \n\nfrom nltk.corpus import stopwords\n\n# lets download the stopwords\nnltk.download('stopwords')\n\n# corpus contains stopwords from all the languages\n# lets see english stopwords \nenglish_stopwords = stopwords.words('english')\n\n\n# simple way to print these stopwords using string formatting\n\n', '.join(english_stopwords) # it gonna print slightly nicer","metadata":{"id":"2ri2HJBeigpN","outputId":"88da56f0-5713-4aaa-97d1-d419af30bee1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so now how do we remove these stopwords\n# -> by simply defining the function\n# now one more problem here : all the stopwords are in smaller but in data there may not be in smaller sometimes capital as well\n# so we use lower() \n\ndef remove_stopwords(tokens):\n  return [word for word in tokens if word.lower() not in english_stopwords]","metadata":{"id":"d4Hdf18diglr"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets remove stopwords from q1 \n\nq1_stp = remove_stopwords(q1_tokenize)\n# lets compare \nprint('word: ',q1)\nprint('after removing stopwords: ', q1_stp)","metadata":{"id":"ddX3mwdol-wb","outputId":"d4212e60-ae51-425a-b533-fb2260c8adfa"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so word : do, you, have, an, how, to, and, not , are all removed","metadata":{"id":"1NHVChVil-sy"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets remove stopwords from q2\n\nq2_stp = remove_stopwords(q2_tokenize)\n# lets compare \nprint('word: ',q2)\nprint('after removing stopwords: ', q2_stp)","metadata":{"id":"_LrdAw4Cl-qL","outputId":"dbc46a0b-2566-4ec1-db13-8a9f4050be58"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# once again we can see lots of words are gone\n#\n# it is going to reduce our vocabulary pretty significantly ","metadata":{"id":"wrkbRzocl-II"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Stemming\n\nstemming is process of turning things like go, going, gone into root word go.\n\nor bird, birds into bird.\n\n\nwhy this ? because , in terms of counting frequencies these words have exact effort.\n\nas we are creating bag of word hence we are decreasing vocabulary and counting frequency of the words\n\n","metadata":{"id":"ri2otdGXnXGA"}},{"cell_type":"code","source":"# we have several stemming libraries\n# PorterStemmer, SnowballStemmer\n# lets try SnowballStemmer\n\nfrom nltk.stem.snowball import SnowballStemmer\n\nstemmer = SnowballStemmer(language = 'english')\n\nstemmer.stem('going')","metadata":{"id":"jzniOO9JneJg","outputId":"72582eaa-52d0-4869-8a57-839514678e3b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer.stem('loudly')","metadata":{"id":"-wS1gHP2neGc","outputId":"08f1167d-8a0a-499c-dd9d-83593b11bf62"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_stem = [stemmer.stem(word) for word in q1_stp]\n# lets compare after stemming\nprint(q1_stp)\nprint(q1_stem)","metadata":{"id":"0hqQynKQneC0","outputId":"4f1595f9-b0f3-40ae-c22b-1285a15e63f9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# so it returned root words ","metadata":{"id":"gicNTlhtnbcE"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q2_stem = [stemmer.stem(word) for word in q2_stp]\n# lets compare after stemming\nprint(q2_stp)\nprint(q2_stem)","metadata":{"id":"CdVZKTLMptbr","outputId":"3861e476-2b65-4c22-975c-a913c05d048f"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Lemmatization\n\nas stemming would reduce words to roots  like in\n\nlove, loving, lovable to lov which makes no sennse\n\nbut what lemmatization would do is \n\nlove, loving, lovable -> love\n\nremove ing, able, etc and add e in the end\n\n\nhow would it know to how to add 'e', it will look at the dictionary internally .\n\n\nlemmatization is not generally used while creating bag of words model cause it utilizes the dictionary internally which can be slow fairly and it can take a lot of memory.","metadata":{"id":"7st1-l2Zp_WV"}},{"cell_type":"markdown","source":"so with that we have completed the text preprocessing.\n\nnow lets create a bag of model.","metadata":{"id":"G18DNO4CsU5D"}},{"cell_type":"markdown","source":"## Implement the Bag of Words Model.\n\nOutline:\n\n1. create a vocabulary using count vectorizer\n2. transform text to vectors using count vectorizer\n3. configure text preprocessing in count vectorizer ","metadata":{"id":"JbcAcAWqsOQ7"}},{"cell_type":"markdown","source":"##### Create a vocabulary using count vectorizer","metadata":{"id":"s2FLD1y4uMOp"}},{"cell_type":"code","source":"sample_df","metadata":{"id":"g0jyIHoisNaK","outputId":"57faa771-b563-473b-a1d1-f3beea9bd8ad"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# As we sampled out 100_000 questions in sample_df\n#\n# 100_000 is large data so lets first work with small sample ie 5 questions\n\nsmall_sample = sample_df[:5]\nsmall_sample.question_text.values","metadata":{"id":"hvHvpjQMsG8y","outputId":"10bd1d9c-b8fc-4651-ac64-3962087642d0"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see small_df itself\nsmall_sample","metadata":{"id":"-6odtvdqt3kd","outputId":"465349c3-91b4-434a-fec9-be196549956d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import count vectorizer from sklear\n\nfrom sklearn.feature_extraction.text import CountVectorizer\n\n# lets instantiate CountVectorizer\nsmall_vect = CountVectorizer()","metadata":{"id":"jOeAFogEt3dr"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_vect.fit(small_sample.question_text)","metadata":{"id":"p-ZQLXVXvOFg","outputId":"2306f7fd-28f3-4c7a-fb22-18f506650409"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"small_vect.vocabulary_","metadata":{"id":"Jl9k4nzXvOB_","outputId":"47a905ea-1f44-4df3-ca75-35e0ae646b99"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the list of words\nsmall_vect.get_feature_names_out()","metadata":{"id":"XAHBPZc7vN_Q","outputId":"60ebbbf8-f9d4-4ed4-e51a-2ef398dc5bc6"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Transform Documents into vectors\n","metadata":{"id":"FruKBsGcwD-Z"}},{"cell_type":"code","source":"vectors = small_vect.transform(small_sample.question_text)\nvectors","metadata":{"id":"EI9kbdH9wKBs","outputId":"972a56a0-b232-4354-85fb-77b027cac942"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since there are lots of words with 0.\n# to store them efficiently, numpy internally uses something called sparse matrix\n# to see this sparse matrix use toarray(), but be careful while doing this coz if the matrix is too large it can run out of  memory\n# but in this case we have small sample so its ok\n\nvectors.toarray()","metadata":{"id":"sc8Xh7KzwJ2-","outputId":"1be470f7-e992-4bb4-93cc-2d376fbb2966"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we can check shapes as well\n\nvectors.shape","metadata":{"id":"1GllUt2syBMZ","outputId":"9821928a-5d1c-475a-d7c6-504bf10029fa"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Configureing count vectorizer parameters","metadata":{"id":"5VPSq4VRFKrB"}},{"cell_type":"code","source":"def tokenize(text):\n  return [word for word in word_tokenize(text) if word.lower() not in english_stopwords]","metadata":{"id":"oIEw2TZMFnW7"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lets check the function\n\ntokenize('what is the real deal here?')","metadata":{"id":"dlry5VVJF-Q0","outputId":"1609a9fb-83ea-4a66-a5d0-4db2fec3215b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer = SnowballStemmer(language='english')","metadata":{"id":"5b2Pia7bGnCK"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize(text):\n  return [stemmer.stem(word) for word in word_tokenize(text)]","metadata":{"id":"UrdMPXn_G8DE"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = CountVectorizer(lowercase=True,\n                             tokenizer = tokenize,\n                             stop_words = english_stopwords, # here we can use either tokenizer or stop words as we have already customized them\n                             max_features= 1000) ","metadata":{"id":"VrRiYAUoFSyT"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we want to perform vectorizer on entire sample_dataset rather than small_sample\n\n\n# it gonna print the time it took to execute\n\nvectorizer.fit(sample_df.question_text)","metadata":{"id":"sG3gc1iLFSuz","outputId":"c0945452-762a-4100-97b3-cdd6ebbe05cc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now it has learned the vocabulary\n# no we can check the vocabulary\n\nlen(vectorizer.vocabulary_)","metadata":{"id":"1tO1ji7TFSn5","outputId":"db8571fd-46c6-414d-a55c-d13433d1f03d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check vocabulary by seeing 100 words\nvectorizer.get_feature_names_out()[:100]","metadata":{"id":"hGKzBsKVIh67","outputId":"d38ed403-00c9-4303-883f-0806336aab9e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we are finally ready to transform our text questions into vectors\ninputs = vectorizer.transform(sample_df.question_text)","metadata":{"id":"pPnrnSJVIhsq","outputId":"c5f0fe86-12ff-4bbd-a369-a6925132afdf"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.shape","metadata":{"id":"FkfG-X4nJdWt","outputId":"62006194-3a0a-4829-885a-5b3a60858334"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets do this for test dataframe","metadata":{"id":"l_UD6Rn2JdTY"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train ML Models for Text Classification.","metadata":{"id":"fhRayyALsH58"}},{"cell_type":"code","source":" test_df","metadata":{"id":"rjSFHtqzs-0b","outputId":"c83049b0-45dc-4cb3-cbbe-4b5ca86f079f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_inputs = vectorizer.transform(test_df.question_text)","metadata":{"id":"l45c1zFGs-3S","outputId":"d996ee89-0d40-44ad-e771-88c855d0f989"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ML Model for text classification\n\nOutline:\n\n*   Create a training and a validation set\n*   Train a Decision Tree Classifier model\n*   Make prediction on training, validation and test data\n\n\n\n","metadata":{"id":"ePu8tn2r9fne"}},{"cell_type":"markdown","source":"Splitting into Training and Validation set","metadata":{"id":"mxs8J2GT-x-a"}},{"cell_type":"code","source":"sample_df\n\n# since our whole dataset is too large so we are using a sample ","metadata":{"id":"wE6Xzg4Ps-60","outputId":"3f10c790-03fe-4bf6-e5cc-74f563b0a31d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs.shape\n\n# inputs is we have vectorized sample_df.question_text using count vectorizer in above cells","metadata":{"id":"I6YS_k7Ws_Q2","outputId":"8b6b30dd-f564-455b-afd2-59317e7aa5ec"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since decision tree classifier does not take the sparse array as input.\n# as the count vectorizer returns sparse array as output\n# so converting the inputs to numpy arrays\n\ninputs = inputs.toarray()","metadata":{"id":"8jQABvx6EWFh"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n\ntrain_inputs,val_inputs, train_targets , val_targets = train_test_split(inputs, \n                                                                        sample_df.target,\n                                                                        test_size=0.3,\n                                                                        random_state=56)","metadata":{"id":"u_snmDcW_HIS"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train_inputs shape: ', train_inputs.shape)\nprint('train_targets shape: ', train_targets.shape)\nprint('val inputs shape: ', val_inputs.shape)\nprint('val targets shape: ', val_targets.shape)","metadata":{"id":"z8OWe4BfAT3n","outputId":"cbab67b2-bc3c-4232-fc6d-6b84e0bf3684"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train Decision Tree classifier","metadata":{"id":"ecmJV1HQBIEK"}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier","metadata":{"id":"W3BXNx3dA_t-"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = DecisionTreeClassifier(max_features = 1000)","metadata":{"id":"ABiWboM1BSHl"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_inputs, train_targets)","metadata":{"id":"atGwm_PzCJsp","outputId":"d9e3cfcf-fb89-4a4f-f4e9-0fb282ae0fb9"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Make Predictions","metadata":{"id":"E-E6xc3nJS2t"}},{"cell_type":"code","source":"train_preds = model.predict(train_inputs)","metadata":{"id":"nUfW8WqvDJS0"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets see the accuracy of the prediction of our model\n\nfrom sklearn.metrics import accuracy_score\n\naccuracy_score(train_targets, train_preds)","metadata":{"id":"fy803wKbJiPm","outputId":"f91e896e-376d-4ec2-f16f-c3fb5a570f32"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since this competition is scored using F1 score so lets see the f1 score\n\nfrom sklearn.metrics import f1_score\n\nf1_score(train_targets, train_preds)","metadata":{"id":"oA9KZp_CKBUF","outputId":"740f5ac7-8bb9-4df5-e9a8-6c579b94cc4a"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"lets compare it with dummy/ random predictions","metadata":{"id":"XK9DJ9pDLUFD"}},{"cell_type":"code","source":"import numpy as np\nrandom_preds = np.random.choice((0,1), len(train_targets))","metadata":{"id":"qvBU9HEPKhnr"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_score(train_targets, random_preds)","metadata":{"id":"KCzvaiuqL2qg","outputId":"41c9a664-fc72-4c10-e496-07bada3b8cef"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets check the accuracy on validation set","metadata":{"id":"qcEdJdt0MJQS"}},{"cell_type":"code","source":"val_preds = model.predict(val_inputs)","metadata":{"id":"zpTcpjZZME1G"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('accuracy: ', accuracy_score(val_targets, val_preds))\nprint('f1 score: ', f1_score(val_targets, val_preds))","metadata":{"id":"ZnFUK6uvMXQu","outputId":"236e4cb6-71aa-482b-d84d-98e7d817a983"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets optimize our model","metadata":{"id":"otN3LVm4NBdC"}},{"cell_type":"code","source":"# lets check for max_d\nfor max_d in range(1,21):\n  model = DecisionTreeClassifier(max_depth=max_d, random_state=87)\n  model.fit(train_inputs, train_targets)\n  print('The Training Accuracy for max_depth {} is:'.format(max_d), model.score(train_inputs, train_targets))\n  print('The Validation Accuracy for max_depth {} is:'.format(max_d), model.score(val_inputs,val_targets))\n  print('')","metadata":{"id":"3tJn6NYVMud2","outputId":"39a46b6a-0e89-4056-fa84-58878ed9ff7c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check for max leaf nodes\nfor min_samp in range(2,6):\n  model = DecisionTreeClassifier(max_depth=20,min_samples_split=min_samp, random_state=87)\n  model.fit(train_inputs, train_targets)\n  print('The Training Accuracy for min_sample_split {} is:'.format(min_samp), model.score(train_inputs, train_targets))\n  print('The Validation Accuracy for min_sample_split {} is:'.format(min_samp), model.score(val_inputs,val_targets))\n  print('')","metadata":{"id":"aWPiVAH9NGHv","outputId":"84f697ac-9c53-4044-8952-c66966bd0ba0"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"lets make prediction for test set\n- we chose the parameter max_depth = 20 for final predictions.","metadata":{"id":"2U2MKLEiNHoP"}},{"cell_type":"code","source":"model = DecisionTreeClassifier(max_depth = 20)","metadata":{"id":"bw3L-6PBywPE"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" model.fit(train_inputs, train_targets)","metadata":{"id":"yhCaWsOAy19l","outputId":"0ca1e242-4855-4632-c2d6-8bd2329643ca"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = model.predict(test_inputs)","metadata":{"id":"EKCm0QdMNRBj"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"id":"i_9TAMNkOT-s","outputId":"71cb8c29-9dd5-4b22-e932-de0b5b5bb480"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.prediction = test_preds","metadata":{"id":"fNR4L6AEzFdM"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"id":"8pObm18FzqxP","outputId":"157f7d7b-0890-446b-c973-414c53fa653d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=None)","metadata":{"id":"w8PA1N3Jz0Fn"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"id":"34jt41KOz6Pj","outputId":"146d131c-a8ee-4c8f-e8aa-4787d210769b"},"execution_count":null,"outputs":[]}]}