{"cells":[{"metadata":{"trusted":true,"_uuid":"284d27a3d8f53b62fda20dc668ae007ad65c7356"},"cell_type":"code","source":"# https://towardsdatascience.com/multi-class-text-classification-model-comparison-and-selection-5eb066197568","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"beca836dfd74e539b935632e97abe679b9a9ad29"},"cell_type":"markdown","source":"Model selection is the task of selecting a statistical model from a set of candidate models, given data. In the simplest cases, a pre-existing set of data is considered. Given candidate models of similar predictive or explanatory power, the simplest model is most likely to be the best choice."},{"metadata":{"_uuid":"c1097b08c3269df191c936e58e2a604b405d7602"},"cell_type":"markdown","source":"The data is available in Google BigQuery that can be downloaded from here. The data is also publicly available at this Cloud Storage URL: https://storage.googleapis.com/tensorflow-workshop-examples/stack-overflow-data.csv."},{"metadata":{"trusted":true,"_uuid":"eb6cccb71d4f81b2a4686c4891644dacb79d6faf"},"cell_type":"code","source":"import logging\nimport pandas as pd\nimport numpy as np\nfrom numpy import random\nimport gensim\nimport nltk\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.metrics import accuracy_score, confusion_matrix\nimport matplotlib.pyplot as plt\nfrom nltk.corpus import stopwords\nimport re\n# from bs4 import BeautifulSoup\n\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6c42c59d0bac1a403818e5a0a8eb4fff977ef3d"},"cell_type":"code","source":"df_train = pd.read_csv('../input/train.csv')\n# df = df[pd.notnull(df['tags'])]\ndf_train.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"be984b3dd605ba0016cd46b6ba9e2802634c3011"},"cell_type":"code","source":"# df['post'].apply(lambda x: len(x.split(' '))).sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"48f2cab7e1e0f837afde637381626d7ca18a2696"},"cell_type":"markdown","source":"We have over 10 million words in the data."},{"metadata":{"trusted":true,"_uuid":"130887e92fcbe9162744adc796c1d96569f647c4"},"cell_type":"code","source":"plt.figure(figsize=(10,4))\ndf_train.Category.value_counts().plot(kind='bar');","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9c281b24fa078b20e95607f3c4f6b7704cb60aec"},"cell_type":"markdown","source":"The classes are NOT very well balanced."},{"metadata":{"_uuid":"812b43ab2a2c283f1f9e5b75982a8f89a906ae01"},"cell_type":"markdown","source":"### BOW with keras"},{"metadata":{"trusted":true,"_uuid":"ad78ed5cfddac345fd2937b2570c4cb867ea27ef"},"cell_type":"code","source":"import itertools\nimport os\n\n%matplotlib inline\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nfrom sklearn.preprocessing import LabelBinarizer, LabelEncoder\nfrom sklearn.metrics import confusion_matrix\n\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Activation, Dropout\nfrom keras.preprocessing import text, sequence\nfrom keras import utils","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b70015647894b0dbcaa9498b3cf1b08bedf499ec"},"cell_type":"code","source":"from keras import backend as K\nK.tensorflow_backend._get_available_gpus()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5daa9ecd60b63c6396b1d8221463162aee07b1a7"},"cell_type":"code","source":"train_posts, test_posts, train_tags, test_tags = train_test_split(df_train['title'], \n                                                                  df_train['Category'], \n                                                                  test_size=0.001, \n                                                                  random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aacbc2de65d10325fc296397ba07fcc349072b0d"},"cell_type":"code","source":"max_words = 120\ntokenize = text.Tokenizer(num_words=max_words, char_level=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"180e936d583d214beb119636f6c63faa452fa570"},"cell_type":"code","source":"tokenize.fit_on_texts(train_posts) # only fit on train\nx_train = tokenize.texts_to_matrix(train_posts)\nx_test = tokenize.texts_to_matrix(test_posts)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4362e467fe7729971b54ae05648b517de9a998b8"},"cell_type":"code","source":"encoder = LabelEncoder()\nencoder.fit(train_tags)\ny_train = encoder.transform(train_tags)\ny_test = encoder.transform(test_tags)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da7d107d49220ae22e71225d1c75c95b37943e27"},"cell_type":"code","source":"num_classes = np.max(y_train) + 1\ny_train = utils.to_categorical(y_train, num_classes)\ny_test = utils.to_categorical(y_test, num_classes)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc98d68c715b8e910a9e1772d8a7cbf7bfa8694b"},"cell_type":"code","source":"print('x_train shape:', x_train.shape)\nprint('x_test shape:', x_test.shape)\nprint('y_train shape:', y_train.shape)\nprint('y_test shape:', y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67db32bb94114ccb38a7d939a8fb5240aa36c05d"},"cell_type":"code","source":"batch_size = 32\nepochs = 2","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2821ec978beb69dfc75e1a12275e6beead58eca2"},"cell_type":"code","source":"# Build the model\nmodel = Sequential()\nmodel.add(Dense(512, input_shape=(max_words,)))\nmodel.add(Activation('relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(num_classes))\nmodel.add(Activation('softmax'))\n\nmodel.compile(loss='categorical_crossentropy',\n              optimizer='adam',\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b614704389dfdad5e89ca29c57dd1c321cfc96b"},"cell_type":"code","source":"history = model.fit(x_train, y_train,\n                    batch_size=batch_size,\n                    epochs=epochs,\n                    verbose=1,\n                    validation_split=0.1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eeb54484ef3feb0db67c4b54d06fb5ae30508f83"},"cell_type":"code","source":"score = model.evaluate(x_test, y_test,\n                       batch_size=batch_size, verbose=1)\nprint('Test accuracy:', score[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7b0ed8ff5e0bfb97c30496fb5d68722a04c871b0"},"cell_type":"code","source":"df_public = pd.read_csv('../input/test.csv')\ndf_public.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b09346dc88402a0ee14440be4d50092de2ddf5db"},"cell_type":"code","source":"x_public = tokenize.texts_to_matrix(df_public[\"title\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f0d67fdf3666150f7b2805a0c5cc3b25b853872"},"cell_type":"code","source":"preds = model.predict(x_public, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8aac0f67f1f51508516e9df44558ff6a7e6e9d1b"},"cell_type":"code","source":"df_public['Category'] = [np.argmax(pred) for pred in preds]\ndf_submit = df_public[['itemid', 'Category']].copy()\ndf_submit.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7deceaf752c7aa6e3a91c067b4e2dfe890a10b66"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a52df1a01b0e2b7c4f59e77816e87666168a5fa0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}