{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport os\nimport time\nimport datetime\nimport gc\nimport random\nimport re\nimport operator\n\nfrom tqdm import tqdm\n\nfrom sklearn.model_selection import StratifiedKFold,KFold\nfrom sklearn.metrics import f1_score,precision_score,recall_score\n\nimport torch\nimport torch.nn as nn\nfrom torch.autograd import Variable\nfrom torch.utils.data import DataLoader,TensorDataset,Dataset\nfrom torch.nn.utils.rnn import pack_padded_sequence, pad_packed_sequence\nfrom torch.optim.optimizer import Optimizer\n\nfrom keras.preprocessing.text import Tokenizer,text_to_word_sequence\nfrom keras.preprocessing.sequence import pad_sequences\n\ndef seed_everything(SEED=42):\n    random.seed(SEED)\n    np.random.seed(SEED)\n    torch.manual_seed(SEED)\n    torch.cuda.manual_seed(SEED)\n    torch.cuda.manual_seed_all(SEED)\n    torch.backends.cudnn.deterministic = True\n    os.environ['PYTHONHASHSEED']=str(SEED)\n    # torch.backends.cudnn.benchmark = False\n\ndef init_func(worker_id):\n    np.random.seed(SEED+worker_id)\n\n    \ntqdm.pandas()\nSEED=42\nseed_everything(SEED=SEED)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"afcf841e7c97d13f36410caef67c2a7d8d197668"},"cell_type":"markdown","source":"## EMBEDDINGS"},{"metadata":{"trusted":true,"_uuid":"9b36d9656bf10eedc0e52f655ba2c78926ddc095"},"cell_type":"code","source":"%%time\ndef load_embed(file):\n    def get_coefs(word,*arr): \n        return word, np.asarray(arr, dtype='float32')\n    \n    if file == '../input/quora-insincere-questions-classification/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec':\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file) if len(o)>100)\n    else:\n        embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(file, encoding='latin'))\n        \n    return embeddings_index\n\nglove = '../input/quora-insincere-questions-classification/embeddings/glove.840B.300d/glove.840B.300d.txt'\n\nprint(\"Extracting GloVe embedding\")\nembeddings_dict_glove= load_embed(glove)\nprint(\"Number of embeddings loaded:\",len(embeddings_dict_glove))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8d55ca674423c1960ec555a0f97d31f696415d0e"},"cell_type":"markdown","source":"## DATA"},{"metadata":{"trusted":true,"_uuid":"1e441f5b351901e0360e667aecafc4f33c906301"},"cell_type":"code","source":"path=\"../input/\"\ntrain=pd.read_csv(path+\"hackereathmlinterntest/756269323c1011e9/dataset/hm_train.csv\")\ntest=pd.read_csv(path+\"hackereathmlinterntest/756269323c1011e9/dataset/hm_test.csv\")\nsample=pd.read_csv(path+\"hackereathmlinterntest/756269323c1011e9/dataset/sample_submission.csv\")\n\nprint(train.shape,test.shape,sample.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"16f61e0804e3a037ab188c09ec3ada7ae85bbd4b"},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"266969748ea1e6a91a4195d0daf11103be452643"},"cell_type":"markdown","source":"## DROPPING THE DUPLICATES"},{"metadata":{"_uuid":"058a69b6dce196a106add24c69dedac7f36f89cc"},"cell_type":"markdown","source":"Here we will check the text."},{"metadata":{"trusted":true,"_uuid":"53d005c718f16f53e43fd05b56895bc8c4cb9608"},"cell_type":"code","source":"print(\"For ID\",train['hmid'].nunique()==train.shape[0])\nprint(\"For text\",train['cleaned_hm'].nunique()==train.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"83bd11b2e2b0964c9f93a92d3de4bf9ea8a8aa05"},"cell_type":"markdown","source":"So there are some rows in the dataset which have same text and different ID. So the rows are duplicated only with ID changed. So we need to remove those duplicated rows in the train dataset. Let us see how is the case in test dataset."},{"metadata":{"trusted":true,"_uuid":"cf2e39108137a29b8102d8a673a3c44db1b7e91c"},"cell_type":"code","source":"print(\"For ID\",train['hmid'].nunique()==train.shape[0])\nprint(\"For text\",train['cleaned_hm'].nunique()==train.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"70b91b4267b136f3d9e57b787df7a25a01af360d"},"cell_type":"markdown","source":"So even in the test data we have duplicated rows , but we have to just the predict the category, so it's fine here. Now I will remove duplicated rows in the train dataset. As ID is not at all used even in the future, so I am dropping the ID from the train dataset."},{"metadata":{"trusted":true,"_uuid":"06ec2a6043251546a8492db9ca61a2c0e2cfd6c2"},"cell_type":"code","source":"train_id=train['hmid']\ntrain.drop(columns=['hmid'],inplace=True)\nprint(train.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be824af9c125fc3f6b19e4eb3c6d0f1a3462fdbd"},"cell_type":"code","source":"# dropping the duplicates \ntrain.drop_duplicates(inplace=True)\nprint(train.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"87f79de0ecb01333ae056e98bb6a7e6f177b5196"},"cell_type":"markdown","source":"So we can see easily that from 60321 to the examples dropped to 58682. That's a decrease of 1639 examples."},{"metadata":{"_uuid":"7242689d3849560d91afa4b55afaa5d17d1a7bfb"},"cell_type":"markdown","source":"## TARGET VARIABLE\n\nDistribution of target variable."},{"metadata":{"trusted":true,"_uuid":"851fd130c35cc9ebe24b8f3718c541d4aac51dce"},"cell_type":"code","source":"sns.countplot(train['predicted_category'])\nplt.xticks(rotation='90')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bbc05bd6b9a389997ec1abd5a377ff575f81fe9a"},"cell_type":"code","source":"target_info=train['predicted_category'].value_counts().reset_index()\ntarget_info['percentage']=(target_info['predicted_category']/train.shape[0]*100).astype(str)+\" %\"\ntarget_info","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cfab399b9b7b24f7b4d4a003f7cb44949af5457"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"77dc606fe6b29636f578c4a05a92ed178b5e3ebb"},"cell_type":"markdown","source":"The dataset looks highly imbalanced. There are 7 categories. "},{"metadata":{"_uuid":"d8d19ceb355bb76078aad07ce19071cd5fd56692"},"cell_type":"markdown","source":"## NUMBER OF SENTENCES"},{"metadata":{"trusted":true,"_uuid":"6417bde289cbfa426e24d08d01f87a4b96fb6e4c"},"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.subplot(131)\nsns.distplot(train['num_sentence'],kde=False)\nplt.title(\"Train Distribution\")\nplt.yscale(\"log\")\nplt.subplot(132)\nsns.boxplot(train['predicted_category'],y=train['num_sentence'])\nplt.xticks(rotation=\"90\")\nplt.subplot(133)\nsns.distplot(test['num_sentence'],kde=False)\nplt.title(\"Test Distribution\")\nplt.yscale(\"log\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"410773e3ca99ba99cdb87b9eab37f3a440ba9919"},"cell_type":"markdown","source":"So there are statments with number of sentences nearly 60 (which is very high). And you can see from the other graph that the sentences which have 60 sentences are mostly belong to \"affection\". in test data the num of sentences are even more with approx 70 sentences."},{"metadata":{"trusted":true,"_uuid":"8d8df3b4ef4555cc0fa341b38c07c5eab7e6ef1e"},"cell_type":"markdown","source":"## REFLECTION PEROID\n\nIt represents the time of happiness. This variable only takes two values 24h and 3m. Let us see the distribution in train and test and also the relation with target."},{"metadata":{"trusted":true,"_uuid":"a9e3f038e4aab8bf32133f659b07ae5c67ad187b","scrolled":true},"cell_type":"code","source":"plt.figure(figsize=(15,5))\nplt.subplot(131)\nsns.countplot(train['reflection_period'])\nplt.title(\"Train Distribution\")\nplt.subplot(132)\nsns.countplot(train['reflection_period'],hue=train['predicted_category'])\nplt.subplot(133)\nsns.countplot(test['reflection_period'])\nplt.title(\"Test Distribution\")\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"258a002c615af23b33ee78c4001ab7724ff568df"},"cell_type":"markdown","source":"Number of 3m are more in test data than in train data."},{"metadata":{"_uuid":"cab5cc1dfd0539696393db369a93b8e52ef50bf8"},"cell_type":"markdown","source":"## WORD LENGTH\n\n   In this we will examin the word length of the sentences."},{"metadata":{"trusted":true,"_uuid":"0498a7e8a95a5affa8c392271f7ffebd40e6398a"},"cell_type":"code","source":"train['num_words']=train['cleaned_hm'].apply(lambda x:len(x.split()))\ntest['num_words']=test['cleaned_hm'].apply(lambda x:len(x.split()))\nprint(train.shape,test.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"40b78c09a8d1d3cbfc3487986b7067d523ab0415"},"cell_type":"code","source":"print(\"The average word length in train is\",train['num_words'].mean(),\n                      \"and in test is\",test['num_words'].mean())\n\nplt.figure(figsize=(12,5))\nplt.subplot(121)\nsns.distplot(train['num_words'],kde=False)\nplt.yscale(\"log\")\nplt.title(\"Train Distribution\")\nplt.subplot(122)\nsns.distplot(test['num_words'],kde=False)\nplt.yscale(\"log\")\nplt.title(\"Test Distribution\")\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fb26f7ef2233ebc65011e0ee7f68f64a5438eaae"},"cell_type":"markdown","source":"The word length looks same. But there are some sentences for which the length is 1200 which is very high for RNN,LSTM. But in train and test , the average word length is 19 and 20. As there are outliers , it's better we see the median length also. "},{"metadata":{"trusted":true,"_uuid":"7c42ac84e63e527d8c0533fc7d0a8a1c23358f24"},"cell_type":"code","source":"print(\"The median word length in train is\",train['num_words'].median(),\n                      \"and in test is\",test['num_words'].median())","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"22744cbf0f6942f9cecc353ecea62dbe96c5a2cb"},"cell_type":"markdown","source":"So median word length is 14 for train and 13 for test."},{"metadata":{"trusted":true,"_uuid":"daaf513e702fd5fe33d4a14a7064e3649add3198"},"cell_type":"markdown","source":"## TEXT"},{"metadata":{"trusted":true,"_uuid":"2829dfba45c49c2db5f9901f0879314cea793b75"},"cell_type":"code","source":"def build_vocab(sentences,verbose=True):\n    vocab={}\n    for sentence in tqdm(sentences,disable=(not verbose)):\n        for word in sentence:\n            try:\n                vocab[word]+=1\n            except KeyError:\n                vocab[word]=1\n                \n    print(\"Number of words found in vocab are\",len(vocab.keys()))\n    return dict(sorted(vocab.items(), key=operator.itemgetter(1))[::-1])\n\ndef sen(x):\n    return x.split()\n\ndef check_coverage(vocab,embeddings_dict):\n    # words that dont have embeddings\n    oov={}\n    # stores words that have embeddings\n    a=[]\n    i=0\n    k=0\n    for word in tqdm(vocab.keys()):\n        if embeddings_dict.get(word) is not None:                    # implies that word has embedding\n            a.append(word)\n            k=k+vocab[word]\n        else:\n            oov[word]=vocab[word]\n            i=i+vocab[word]\n    \n    print(\"Total embeddings found in vocab are\",len(a)/len(vocab)*100,\"%\")\n    print(\"Total embeddings found in text are\",k/(k+i)*100,\"%\")\n    sorted_x = sorted(oov.items(), key=operator.itemgetter(1))[::-1]\n    return dict(sorted_x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ed00801a7f54eb7fa575d6072bd1cb866a136b0a"},"cell_type":"markdown","source":"The code for the analysis written below is removed.\n\nAs you can see the words that dont have embedding are in compact form an not in loose form. For example if haven't was written as have not then there is no problem finding a embedding. As haven't and have not doesn't change the meaning of the sentence we can replace them easily. The mappings are defined below."},{"metadata":{"trusted":true,"_uuid":"bd745de9ef98ce11a1e18689a83e3493bbbbe5d6","scrolled":true},"cell_type":"code","source":"puncts=[',','.','!','$','(',')','%','[',']','?',':',\";\",\"#\",'/','\"',\"'\",\"-\",\"|\",'*']\n\ncontraction_mapping={\"haven't\":\"have not\",\"hadn't\":\"had not\",\"wasn't\":\"was not\",\"he's\":\"he is\",\n                     \"couldn't\":\"could not\",\"she's\":\"she is\",\"i'm\":\"i am\",\"we've\":\"we have\",\n                     \"wouldn't\":\"would not\",\"That's\":\"That is\",\"we're\":\"we are\",\"isn't\":\"is not\",\n                     \"hasn't\":\"has not\",\"they're\":\"they are\",\"She's\":\"She is\",\"He's\":\"He is\",\"weren't\":\"were not\",\n                    \"there's\":\"there is\",\"i've\":\"i have\",\"you've\":\"you have\",\"We've\":\"We have\",\"we'd\":\"we would\",\n                    \"We're\":\"We are\",\"who's\":\"who is\",\"they'll\":\"they will\",\"what's\":\"what is\",\"she'd\":\"she would\",\n                    \"They're\":\"They are\",\"aren't\":\"are not\",\"shouldn't\":\"should not\",\"There's\":\"There is\",\n                     \"we'll\":\"we will\",\"I`m\":\"I am\",\"You're\":\"You are\",\"i'd\":\"i would\",\"he'll\":\"he will\",\n                    \"they'd\":\"they would\",\"Didn't\":\"Did not\",\"CAN'T\":\"CANNOT\",\"THAT'S\":\"That is\",\"you;ll\":\"you will\",\n                    \"You'll\":\"You will\",\"Can't\":\"cannot\",\"would've\":\"would have\",\"you\\'re\":\"you are\",\"i'll\":\"i will\",\n                    \"DIDN'T\":\"did not\",\"film/theater\":\"film or theater\",\"that\\'s\":\"that is\",\"Let's\":\"Lets\",\"We'd\":\"We would\",\n                    \"They've\":\"They have\",\"she'll\":\"she will\",\"Haven't\":\"Have not\",\"it'll\":\"it will\",\"you'd\":\"you would\",\n                    \"I'VE\":\"I have\",\"I`ve\":\"I have\",\"She'll\":\"She will\",\"It'll\":\"It will\",\"Hadn't\":\"Had not\",\"I\\'ve\":\"I have\",\n                     \"You\\'re\":\"You are\",\"b'day\":\"birthday\",\"DON'T\":\"Do not\",\"it'd\":\"it would\",\"You've\":\"You have\",\n                     \"I'LL\":\"I will\",\"don\\'t\":\"do not\",\"what\\'s\":\"what is\",\"won't\":\"will not\",\n                    \"he'd\":\"he would\",\"I'M\":\"I AM\"}\n\nmisspelled_words={\"fiancA\":\"fiance\",\"couldnat\":\"could not\",\"giftz\":\"gifts\",\n                \"othersa\":\"others\",\"nerous\":\"nervous\",\"wasnat\":\"was not\",\n                \"aIam\":\"I am\",\"arace\":\"a race\",\"10class\":\"10 class\",\n                 \"aHow\":\"How\",\"aWait\":\"Wait\",\"aMumma\":\"Mumma\",\"aWhy\":\"Why\",\n                \"B+\":\"B +\",\"INAGURATION\":\"INAUGURATION\",\"wonat\":\"will not\",\n                \"3+\":\"3 +\",\"Thereas\":\"There is\",\"Letas\":\"Let's\",\n                \"Valentineas\":\"Valentines\",\"genervous\":\"generous\",\n                 \"brotheras\":\"brother's\",\"Dhubai\":\"Dubai\",\n                 \"shiridi\":\"shirdi\",\"PAPPER\":\"PAPER\",\"booka|\":\"book\",\n                 \"aHey\":\"Hey\",\"seek&hide\":\"seek and hide\"}\n\ndef replace_misspelled(x):\n    for word in misspelled_words.keys():\n        x=x.replace(word,misspelled_words[word])\n        \n    return x\n\ndef replace_contraction_mapping(x):\n    for contract in contraction_mapping.keys():\n        x=x.replace(contract,contraction_mapping[contract])\n    \n    return x\n    \ndef replace_puncts(x):\n    for p in puncts:\n        x=x.replace(p,f' {p} ')\n        \n    return x\n\ncleaned_sen=train['cleaned_hm'].progress_apply(replace_misspelled)\ncleaned_sen=cleaned_sen.progress_apply(replace_contraction_mapping)\ncleaned_sen=cleaned_sen.progress_apply(replace_puncts)\nsentences=cleaned_sen.progress_apply(sen)\nvocab=build_vocab(sentences)\noov=check_coverage(vocab,embeddings_dict_glove)\n\n# for index,sen in enumerate(cleaned_sen):\n#     if \"Valentineas\" in sen.split():\n#         print(sen,index)\n#         print(\"\\n\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ab2b355c883dd257aeb5f5e1fff81168ead1ec90"},"cell_type":"code","source":"# cleaning the train data and test data\n\n# replace the misspelled word\ntrain_clean_hm=train['cleaned_hm'].progress_apply(replace_misspelled)\ntest_clean_hm=test['cleaned_hm'].progress_apply(replace_misspelled)\n\n# replace contraction mapping\ntrain_clean_hm=train_clean_hm.progress_apply(replace_contraction_mapping)\ntest_clean_hm=test_clean_hm.progress_apply(replace_contraction_mapping)\n\n# replace punctuations\ntrain_clean_hm=train_clean_hm.progress_apply(replace_puncts)\ntest_clean_hm=test_clean_hm.progress_apply(replace_puncts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f463423831b5c9008c1a70b27eb8e168605643b8"},"cell_type":"markdown","source":"## TOKENIZING AND PADDING"},{"metadata":{"trusted":true,"_uuid":"d4d29f70dc825f1b23ab5ff2e6dc63b1399775ea","scrolled":true},"cell_type":"code","source":"max_words=20000\nmax_len=70\nembed_dim=300\n\n\ntokenizer=Tokenizer(num_words=max_words,filters=None,lower=False)\ntokenizer.fit_on_texts(list(train_clean_hm.apply(sen).values))\n\nprint(\"The length of vocabulary is\",len(tokenizer.word_index))\n\nX=tokenizer.texts_to_sequences(list(train_clean_hm.apply(sen).values))\nX_test=tokenizer.texts_to_sequences(list(test_clean_hm.apply(sen).values))\n\n# padding and truncating\nX=np.array(pad_sequences(X,maxlen=max_len,padding='pre',truncating='pre'))\nX_test=np.array(pad_sequences(X_test,maxlen=max_len,padding='pre',truncating='pre'))\n\n\n# target\ntarget_encoder={\"affection\":0,\"achievement\":1,\n                \"bonding\":2,\"enjoy_the_moment\":3,\n               \"leisure\":4,\"nature\":5,\n               \"exercise\":6}\n\ntarget_decoder=dict(zip(target_encoder.values(),target_encoder.keys()))\n\ny=pd.get_dummies(train['predicted_category'].map(target_encoder)).values\n\nprint(\"Training Shape\",X.shape,y.shape)\nprint(\"Test Shape\",X_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8d5a01811bd5a3d1cdc324069ccf89047497dc0"},"cell_type":"markdown","source":"## EMBEDDINGS MATRIX"},{"metadata":{"trusted":true,"_uuid":"05cef295cd38f4bb3f34e24cace29dd4f43e91d5"},"cell_type":"code","source":"def give_embed_glove(word_index):\n    nb_words=min(max_words,len(word_index)+1)\n    embeddings_matrix_glove = np.zeros((nb_words, embed_dim))\n    for word,index in word_index.items():\n        if index>=max_words:\n            continue\n        # implies that word has embedding\n        if embeddings_dict_glove.get(word) is not None:\n            embeddings_matrix_glove[index]=embeddings_dict_glove.get(word)\n            \n    \n    return embeddings_matrix_glove\n\n\nembeddings_matrix_glove=give_embed_glove(tokenizer.word_index)\nprint(\"The Shape of the glove matrix is\",embeddings_matrix_glove.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2227d997757d41c77063d01673ae07d180fac626"},"cell_type":"markdown","source":"## METRICS"},{"metadata":{"trusted":true,"_uuid":"a0ad8850eba592c3c6a2f136ecd9c04ecb550df6"},"cell_type":"code","source":"def cross_entropy(y_true,y_pred,eps=1e-8):\n    \"\"\"\n    y_true : (m,classes) contaning true values.\n    y_pred : (m,classes) contaning predictions.\n\n    Returns : Cross entropy loss between y_true and y_pred.\n    \"\"\"\n    predictions=np.clip(y_pred,eps,1-eps)\n    m=predictions.shape[0]\n    loss=-np.sum(y_true*np.log(predictions))/m\n    return loss\n\ndef f1(y_true,y_pred,threshold=0.5,eps=1e-8):\n    \"\"\"    \n    y_true : (m,classes) contaning true values.\n    y_pred : (m,classes) contaning predictions.\n    \n    Returns : Average Weighted F1 Score.\n    \"\"\"\n    predictions=np.argmax(y_pred,axis=1)\n    targets=np.argmax(y_true,axis=1)\n    return f1_score(targets,predictions,average=\"weighted\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9e7f39742a73a5b99987a8e405787af13d252bd9"},"cell_type":"markdown","source":"## LOSS"},{"metadata":{"trusted":true,"_uuid":"09e6acf9e3bc0a39d8d87d02b92d7a8bae3f8560"},"cell_type":"code","source":"class CrossEntropyLoss(nn.Module):\n    \"\"\"\n    y_true = (N,C)\n    y_pred = (N,C)\n    Cross Entropy Loss\n    \"\"\"\n    def __init__(self,eps=1e-8):\n        super(CrossEntropyLoss,self).__init__()\n        self.eps=eps\n    \n    def forward(self,y_true,y_pred):\n        y_pred=torch.clamp(y_pred,self.eps,1-self.eps)\n        m=y_pred.shape[0]\n        loss=-torch.sum(y_true*torch.log(y_pred))/m\n        \n        return loss","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d29d2896d58e9a9abbf05cf7118e931224b14add"},"cell_type":"markdown","source":"## CYCLIC LEARNING RATE"},{"metadata":{"trusted":true,"_uuid":"6b212bdf17db05c752ff687c23fe23ac64c9f586"},"cell_type":"code","source":"# code inspired from: https://github.com/anandsaha/pytorch.cyclic.learning.rate/blob/master/cls.py\nclass CyclicLR(object):\n    def __init__(self, optimizer, base_lr=1e-3, max_lr=6e-3,\n                 step_size=2000, mode='triangular', gamma=1.,\n                 scale_fn=None, scale_mode='cycle', last_batch_iteration=-1):\n\n        if not isinstance(optimizer, Optimizer):\n            raise TypeError('{} is not an Optimizer'.format(\n                type(optimizer).__name__))\n        self.optimizer = optimizer\n\n        if isinstance(base_lr, list) or isinstance(base_lr, tuple):\n            if len(base_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} base_lr, got {}\".format(\n                    len(optimizer.param_groups), len(base_lr)))\n            self.base_lrs = list(base_lr)\n        else:\n            self.base_lrs = [base_lr] * len(optimizer.param_groups)\n\n        if isinstance(max_lr, list) or isinstance(max_lr, tuple):\n            if len(max_lr) != len(optimizer.param_groups):\n                raise ValueError(\"expected {} max_lr, got {}\".format(\n                    len(optimizer.param_groups), len(max_lr)))\n            self.max_lrs = list(max_lr)\n        else:\n            self.max_lrs = [max_lr] * len(optimizer.param_groups)\n\n        self.step_size = step_size\n\n        if mode not in ['triangular', 'triangular2', 'exp_range'] \\\n                and scale_fn is None:\n            raise ValueError('mode is invalid and scale_fn is None')\n\n        self.mode = mode\n        self.gamma = gamma\n\n        if scale_fn is None:\n            if self.mode == 'triangular':\n                self.scale_fn = self._triangular_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'triangular2':\n                self.scale_fn = self._triangular2_scale_fn\n                self.scale_mode = 'cycle'\n            elif self.mode == 'exp_range':\n                self.scale_fn = self._exp_range_scale_fn\n                self.scale_mode = 'iterations'\n        else:\n            self.scale_fn = scale_fn\n            self.scale_mode = scale_mode\n\n        self.batch_step(last_batch_iteration + 1)\n        self.last_batch_iteration = last_batch_iteration\n\n    def batch_step(self, batch_iteration=None):\n        if batch_iteration is None:\n            batch_iteration = self.last_batch_iteration + 1\n        self.last_batch_iteration = batch_iteration\n        for param_group, lr in zip(self.optimizer.param_groups, self.get_lr()):\n            param_group['lr'] = lr\n\n    def _triangular_scale_fn(self, x):\n        return 1.\n\n    def _triangular2_scale_fn(self, x):\n        return 1 / (2. ** (x - 1))\n\n    def _exp_range_scale_fn(self, x):\n        return self.gamma**(x)\n\n    def get_lr(self):\n        step_size = float(self.step_size)\n        cycle = np.floor(1 + self.last_batch_iteration / (2 * step_size))\n        x = np.abs(self.last_batch_iteration / step_size - 2 * cycle + 1)\n\n        lrs = []\n        param_lrs = zip(self.optimizer.param_groups, self.base_lrs, self.max_lrs)\n        for param_group, base_lr, max_lr in param_lrs:\n            base_height = (max_lr - base_lr) * np.maximum(0, (1 - x))\n            if self.scale_mode == 'cycle':\n                lr = base_lr + base_height * self.scale_fn(cycle)\n            else:\n                lr = base_lr + base_height * self.scale_fn(self.last_batch_iteration)\n            lrs.append(lr)\n        return lrs","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"74e0d37392602c9f54f19ea4f78122294586edcc"},"cell_type":"markdown","source":"## MODEL"},{"metadata":{"trusted":true,"_uuid":"1ebbb078a7fa8b9a2b5bc6c008a2a9ce286af984"},"cell_type":"code","source":"class Attention(nn.Module):\n    def __init__(self,hidden_dim,max_len):\n        super(Attention,self).__init__()\n        \n        self.hidden_dim=hidden_dim\n        self.max_len=max_len\n        \n        self.tanh=nn.Tanh()\n        self.linear=nn.Linear(in_features=self.hidden_dim,out_features=1,bias=False)\n        self.softmax=nn.Softmax(dim=1)        \n        \n    def forward(self,h):\n        \n        m=self.tanh(h)\n        \n        alpha=self.linear(m)\n        \n        alpha=torch.squeeze(alpha)           # shape of alpha will be batch_size*max_len\n        \n#         print(\"Alpha shape:\",alpha.shape)\n        \n        # softmax(note that softmax is along dimension 1)\n        alpha=self.softmax(alpha)\n        \n        # unsequezzing alpha to get shape as batch_size*max_len*1\n        alpha=torch.unsqueeze(alpha,-1)\n        \n        # we have to define r\n        r=h*alpha\n        \n        # now we have to take sum and shape of r is batch_size*hidden_size\n        r=torch.sum(r,dim=1)\n        \n        return r","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"232297d60795a51c36854d35d397986543778944"},"cell_type":"code","source":"# model params\nhidden_units=64\n\nclass LstmGru(nn.Module):\n    def __init__(self,embeddings_matrix):\n        super(LstmGru,self).__init__()\n\n        self.embedding=nn.Embedding.from_pretrained(torch.Tensor(embeddings_matrix),freeze=True)\n        \n        self.lstm=nn.LSTM(input_size=embed_dim,hidden_size=hidden_units,\n                              bidirectional=True,batch_first=True)\n        \n        # gru output goes to lstm\n        self.gru=nn.GRU(input_size=2*hidden_units,hidden_size=hidden_units,\n                           bidirectional=True,batch_first=True)\n        \n        # lstm attention\n        self.lstm_attention=Attention(2*hidden_units,max_len)\n        \n        # gru attention\n        self.gru_attention=Attention(2*hidden_units,max_len)\n        \n        \n        self.linear1=nn.Linear(in_features=8*hidden_units,out_features=32)\n        self.batch1=nn.BatchNorm1d(32)\n        self.relu1=nn.ReLU()\n        self.drop1=nn.Dropout(0.25)\n        \n        self.linear2=nn.Linear(in_features=32,out_features=7)\n        self.softmax=nn.Softmax(dim=1)\n        \n        \n    def forward(self,X):\n        batch_size=X.shape[0]\n        \n        embeds=self.embedding(X.long())\n        \n        h_lstm,_=self.lstm(embeds)\n        h_gru,_=self.gru(h_lstm)\n        \n        # max pooling over time\n        h_lstm_max,_=torch.max(h_lstm,1)\n        h_gru_max,_=torch.max(h_gru,1)\n        \n        # mean average pooling over time\n        h_lstm_mean=torch.mean(h_lstm,1)\n        h_gru_mean=torch.mean(h_gru,1)\n        \n        # attention\n        h_lstm_attend=self.lstm_attention(h_lstm)\n        h_gru_attend=self.gru_attention(h_gru)\n\n        h=torch.cat((h_lstm_attend,h_gru_attend,h_lstm_max,h_gru_max),dim=1)\n        \n        output=self.relu1(self.batch1(self.linear1(h)))\n        output=self.drop1(output)\n        \n        output=(self.linear2(output))\n        output=self.softmax(output)\n        \n        return output\n    \ndef initialize_model(embeddings_matrix):\n    model=LstmGru(embeddings_matrix)\n    \n    # setting all the dtypes to float\n    model.float()\n    \n    # pushing the code to gpu\n    model.cuda()\n    \n    # params\n    trainable_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\n    \n    print(\"Total trainiable Param's are\",trainable_params)\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5889792b443cb0b0f2e3e3e6534c73b1bf8e849"},"cell_type":"code","source":"initialize_model(embeddings_matrix_glove)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9abba3413b1a2eecb6e814c9e2628b7491d27597"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"554299873b34dcec859dd1d183c790ff0020e03b"},"cell_type":"markdown","source":"## SOME USEFUL FUNCTIONS"},{"metadata":{"trusted":true,"_uuid":"519084d3ac7df8a3830fe50e31d0c782821147c3"},"cell_type":"code","source":"def fit_data(model,optimizer,loss_fn,scheduler=None,\n             train_iterator=None,val_iterator=None,\n            m_train=None,m_val=None,num_classes=None,epochs=None,fname=None):\n    \"\"\"\n        model : pytroch model\n        optimizer : any optimizer from torch.optim\n        m_train : number of training examples\n        m_val : number of validation examples\n        epochs : number of epochs.\n        fname : file name to save the model(it always saves the best)\n        \n        returns\n        best train preds and best val preds(selected based on validation score)\n        and evaluations(which cotains losses,accuracy and f1 score of every epoch)\n    \"\"\"\n    \n    train_loss=[]\n    val_loss=[]\n    train_f1=[]\n    val_f1=[]\n    evals={}\n    \n    best_train_preds=np.zeros((m_train,num_classes))\n    best_val_preds=np.zeros((m_val,num_classes))\n    best_val_f1=0\n    \n    for ep_num in range(epochs):\n        print(\"Epoch\",\"{0}/{1}:\".format(ep_num+1,epochs))\n        \n        # measuring the current time\n        start=datetime.datetime.now()\n        \n        # iterating through batch\n        train_targets=np.zeros((m_train,num_classes))\n        train_preds=np.zeros((m_train,num_classes))\n    \n        # start train_index\n        train_index=0\n        \n        # setting up the model in train\n        model.train()\n    \n        for batch,(X_train,y_train) in enumerate(train_iterator):\n            optimizer.zero_grad()\n            \n            X_train=Variable(X_train.cuda())\n            y_train=Variable(y_train.type('torch.FloatTensor').cuda())\n            \n            y_pred=model.forward(X_train)\n            \n            if scheduler:\n                scheduler.batch_step()\n                \n            loss=loss_fn(y_train,y_pred)\n            \n            # appending the preds\n            train_preds[train_index:train_index+X_train.shape[0]]=y_pred.cpu().detach().numpy().reshape((-1,num_classes))\n            \n            # appending the targets\n            train_targets[train_index:train_index+X_train.shape[0]]=y_train.cpu().detach().numpy().reshape((-1,num_classes))\n            \n            # backprop\n            loss.backward()\n            \n            # update the weights\n            optimizer.step()\n            \n            train_index=train_index+X_train.shape[0]\n            \n            logger=str(train_index)+\"/\"+str(m_train)\n\n            print(logger,end='\\r')\n            \n        # setting up in evaluation mode\n        model.eval()\n        \n        val_targets=np.zeros((m_val,num_classes))\n        val_preds=np.zeros((m_val,num_classes))\n        val_index=0\n        \n        for batch,(X_val,y_val) in enumerate(val_iterator):\n            \n            X_val=Variable(X_val.cuda())\n            y_val=Variable(y_val.type('torch.FloatTensor').cuda())\n            \n            y_pred=model.forward(X_val)\n            \n            # appending the preds\n            val_preds[val_index:val_index+X_val.shape[0]]=y_pred.cpu().detach().numpy().reshape((-1,num_classes))\n            \n            # appending the targets\n            val_targets[val_index:val_index+X_val.shape[0]]=y_val.cpu().detach().numpy().reshape((-1,num_classes))\n            \n            val_index=val_index+X_val.shape[0]\n            \n        # finding the losses and f1 score \n        trainloss=cross_entropy(train_targets,train_preds)\n        valloss=cross_entropy(val_targets,val_preds)\n        \n        trainf1=f1(train_targets,train_preds)\n        valf1=f1(val_targets,val_preds)\n        \n        train_loss.append(trainloss),val_loss.append(valloss)\n        train_f1.append(trainf1),val_f1.append(valf1)\n        \n        # end measuring time \n        end=datetime.datetime.now()\n        \n        print(\"Seconds = \",round((end-start).total_seconds()),end=\" \")\n        \n        print(\"train loss = \",round(trainloss,5),end=\" \")\n        print(\"train f1 = \",round(trainf1,5),end=\" \")\n\n        print(\"val loss = \",round(valloss,5),end=\" \")\n        print(\"val f1 = \",round(valf1,5))\n        \n        if valf1>best_val_f1:\n            print(\"Validation F1 score increased from\",round(best_val_f1,5),\"to\",round(valf1,5),\\\n                                      \"Saving the model at\",fname)\n            \n            torch.save(model.state_dict(),fname)\n            \n            best_val_f1=valf1\n            best_train_preds=train_preds\n            best_val_preds=val_preds\n        print(\"\\n\")\n    \n    # outside of epoch loop\n    evals['train_loss']=train_loss\n    evals['val_loss']=val_loss\n    evals['train_f1']=train_f1\n    evals['val_f1']=val_f1\n    evals['best_val_f1']=best_val_f1\n    \n    return best_train_preds,best_val_preds,evals","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c23baf543a168b163f271e1db5a33ea02557ab2"},"cell_type":"code","source":"def predict_on_test(model,test_iterator,m_test):\n    # model at evaluation mode\n    model.eval()\n    \n    test_preds=np.zeros((m_test,num_classes))\n    test_index=0\n    \n    start=datetime.datetime.now()\n    \n    for batch,X_test in enumerate(test_iterator):\n        X_test=Variable(X_test[0].cuda())\n        \n        y_pred=model.forward(X_test)\n        # appending the preds\n        test_preds[test_index:test_index+X_test.shape[0]]=y_pred.cpu().detach().numpy().reshape((-1,num_classes))\n        \n        test_index=test_index+X_test.shape[0]\n        \n        logger=str(test_index)+\"/\"+str(m_test)\n        \n        if batch<len(test_iterator)-1:\n            print(logger,end='\\r')\n        else:\n            print(logger,end=\" \")\n            \n    end=datetime.datetime.now()\n    print(\"Predictions done on test data in\",round((end-start).total_seconds()),\"seconds\")\n    \n    return test_preds","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7ca173efa2f12ce165f6c0721bd23ab4df089a7d"},"cell_type":"code","source":"print(\"Training Shape\",X.shape,y.shape)\nprint(\"Testing Shape\",X_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c31bdab0968dc2b4c428137820344613acf303e2"},"cell_type":"code","source":"def make_dataset(X_train,y_train,X_val,y_val,batch_size):\n    X_train,y_train=torch.Tensor(X_train),torch.Tensor(y_train)\n    X_val,y_val=torch.Tensor(X_val),torch.Tensor(y_val)\n\n    # train dataset and val dataset contains pair of X and y for each example \n    train_dataset=TensorDataset(X_train,y_train)\n    val_dataset=TensorDataset(X_val,y_val)\n    test_dataset=TensorDataset(X_test)\n\n    # now I will pass this to data loader\n    # shuffle set to true imples for every epoch data is shuffled\n    train_iterator=DataLoader(train_dataset,batch_size=batch_size,shuffle=True)\n    val_iterator=DataLoader(val_dataset,batch_size=batch_size,shuffle=False)\n    \n    return train_iterator,val_iterator","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5339fe4f19994be46b1702c51ca19b99b5cddf9"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9645f522ff31e6abc09a29463df9fb85404a2859"},"cell_type":"markdown","source":"## TRAINING"},{"metadata":{"trusted":true,"_uuid":"1853653516212672ee7a0ec48f1d0af1e9e6ef67","scrolled":true},"cell_type":"code","source":"n_folds=5\nnum_classes=7\nkfold=StratifiedKFold(n_splits=n_folds,shuffle=True,random_state=SEED)\n\nscores=np.zeros((n_folds,))\n\n# oof preds \noof_preds=np.zeros((X.shape[0],num_classes))\n# test preds\ntest_preds=np.zeros((X_test.shape[0],num_classes))\n\n\n# batch size and epochs per fold\nbatch_size=64\nep=[8]*n_folds\n\n# making the test iterator  and for predictions the batch size can be more , so that we can predict fast\nX_test=torch.Tensor(X_test)\ntest_dataset=TensorDataset(X_test)\ntest_iterator=DataLoader(test_dataset,batch_size=256,shuffle=False)\nm_test=X_test.shape[0]\n\n# one point to note is that kfold.split works for labels of form [0,1,1,2]\nfor fold,(train_index,val_index) in enumerate(kfold.split(X,np.argmax(y,axis=1))):\n    X_train,X_val=X[train_index],X[val_index]\n    y_train,y_val=y[train_index],y[val_index]\n    \n    m_train,m_val=X_train.shape[0],X_val.shape[0]\n    \n    print(\"================================= FOLD\",fold+1,\"=============================================\")\n\n    print(\"Training Shape:\",X_train.shape,y_train.shape)\n    print(\"Validation Shape:\",X_val.shape,y_val.shape)\n    \n    train_iterator,val_iterator=make_dataset(X_train,y_train,X_val,y_val,batch_size=batch_size)\n    \n    gc.enable()\n    del X_train,y_train,X_val,y_val\n    gc.collect()\n    \n    model=initialize_model(embeddings_matrix_glove)\n    \n    base_lr=1e-3\n    max_lr=1e-2\n    is_scheduler=True\n    if is_scheduler:\n        step_size=1468                   # 2 times the iteration in an epoch\n        optimizer=torch.optim.Adam(model.parameters(),lr=max_lr)\n        scheduler=CyclicLR(optimizer,base_lr=base_lr,max_lr=max_lr,\n                              step_size=step_size,mode='triangular')\n    else:\n        optimizer=torch.optim.Adam(model.parameters(),lr=base_lr)\n        scheduler=None\n        \n    \n    loss_fn=CrossEntropyLoss()\n    epochs=ep[fold]\n    fname=\"LstmGru\"+str(fold+1)+\".pt\"\n    best_train_preds,best_val_preds,evals=fit_data(model,optimizer,loss_fn,scheduler,train_iterator,val_iterator,\n                                              m_train,m_val,num_classes,epochs,fname)\n    print(\"Loading the model\")\n    model=LstmGru(embeddings_matrix_glove)\n    model.float()\n    model.cuda()\n    model.load_state_dict(torch.load(fname))\n    preds=predict_on_test(model,test_iterator,m_test)\n    \n    gc.enable()\n    del model\n    gc.collect()\n    \n    # saving the oof preds\n    oof_preds[val_index]=best_val_preds\n    \n    # test predictions\n    test_preds=test_preds+preds/n_folds\n    \n    # storing the scores\n    scores[fold]=evals['best_val_f1']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"33fe6c173eacc66e3689f1d76b96aedb56ce47ad"},"cell_type":"code","source":"print(\"The F1 Score on the total data is\",f1(y,oof_preds))\nprint(\"\\n\")\nprint(\"The fold scores are\",scores,\"and the mean is\",np.mean(scores),\"and std is\",np.std(scores))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55998c41c66f7e216c21268704f0fd49a8a52e34"},"cell_type":"markdown","source":"## MAKING THE SUBMISSION FILE"},{"metadata":{"trusted":true,"_uuid":"3a169a018ac2128b344e1de61ed059e0cbdb400f"},"cell_type":"code","source":"sub=pd.DataFrame()\nsub['hmid']=test['hmid']\nsub['predicted_category']=pd.Series(np.argmax(test_preds,axis=1)).map(target_decoder)\nsub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"53790b4de497e727bf34d737b333ef745f23126b"},"cell_type":"code","source":"plt.figure(figsize=(12,5))\nsns.countplot(sub['predicted_category'])\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d6b5129dfdf1072a665b805164d857855327530"},"cell_type":"code","source":"sub.to_csv(\"first_sub.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee78cbee1aec24f667c2f1857ef99d96fb23d28d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b70f04c0f1a8164b17cc6f949ea409143158ab0f"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b55aa8a5fd16ca9b31907f083e50fe34920df83b"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0dacf85ba91ce777c25ef5ec9156e0ca92c765b7"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}