{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0. Informasi\n* NIM: 20210804007\n* Nama: Julianto\n* CMA101 Topik dalam Artificial Inteligence EU101 7673\n* UAS\n* Dataset: Steam User Review - https://store.steampowered.com/ - 2022-07-25 - Total Data: 695,033\n* Class: review_status = 1: Positive Review 0: Negative Review - Total Class: 2\n* Crawler: https://github.com/aesuli/steam-crawler/\n* Reference: https://www.kaggle.com/code/aninditapani/nlp-with-mlp-cnn-and-glove/notebook","metadata":{}},{"cell_type":"markdown","source":"# 1. Load Library","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport string\nimport re\nimport matplotlib.pyplot as plt\n\nfrom keras.preprocessing.text import Tokenizer\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing import sequence\nfrom keras.models import Sequential\nfrom keras.layers import Embedding, Dense, Flatten\nfrom nltk.corpus import stopwords\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:55:25.938771Z","iopub.execute_input":"2022-07-27T15:55:25.940184Z","iopub.status.idle":"2022-07-27T15:55:33.754984Z","shell.execute_reply.started":"2022-07-27T15:55:25.940068Z","shell.execute_reply":"2022-07-27T15:55:33.753779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Load DataSet","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/../input/steam-user-review-20220725/reviews_clean_v2.csv',low_memory=False)\ndf = pd.DataFrame(df) \ndf = df[['review_status','review']]\n\ndf.dropna(how='any',inplace=True)\n\nprint(df.shape)\ndf.head() # What's in there?","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:55:33.757616Z","iopub.execute_input":"2022-07-27T15:55:33.758426Z","iopub.status.idle":"2022-07-27T15:55:44.140244Z","shell.execute_reply.started":"2022-07-27T15:55:33.758376Z","shell.execute_reply":"2022-07-27T15:55:44.138950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Text Preprocessing","metadata":{}},{"cell_type":"code","source":"def text_cleaning(text):\n    '''\n    Converts all text to lower case, Removes special charecters, emojis and multiple spaces\n    text - Sentence that needs to be cleaned\n    '''\n    text = ''.join([k for k in text if k not in string.punctuation])\n    text = re.sub('[^A-Za-z0-9]+', ' ', str(text).lower()).strip()\n    text = re.sub(' +', ' ', text)\n    emoji_pattern = re.compile(\"[\"\n                               u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                               u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                               u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                               u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                               \"]+\", flags=re.UNICODE)\n    text = emoji_pattern.sub(r'', text)\n    return text\n\n\n\ndf[\"review\"] = df[\"review\"].apply(lambda text: text_cleaning(text))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:55:44.141598Z","iopub.execute_input":"2022-07-27T15:55:44.141921Z","iopub.status.idle":"2022-07-27T15:56:53.491133Z","shell.execute_reply.started":"2022-07-27T15:55:44.141871Z","shell.execute_reply":"2022-07-27T15:56:53.490096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STOPWORDS = set(stopwords.words('english'))\ndef remove_stopwords(text):\n    \"\"\"custom function to remove the stopwords\"\"\"\n    return \" \".join([word for word in str(text).split() if word not in STOPWORDS])\n\ndf[\"review\"] = df[\"review\"].apply(lambda text: remove_stopwords(text))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:56:53.493310Z","iopub.execute_input":"2022-07-27T15:56:53.493659Z","iopub.status.idle":"2022-07-27T15:57:01.034009Z","shell.execute_reply.started":"2022-07-27T15:56:53.493627Z","shell.execute_reply":"2022-07-27T15:57:01.032895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_words = 8000 \nt = Tokenizer(num_words=max_words)\nt.fit_on_texts(df['review']) # assigns unique int to all the words\nword_index=t.word_index\nword_index # Can you see that?! It is a dictionary of words to numbers","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-07-27T15:57:01.035319Z","iopub.execute_input":"2022-07-27T15:57:01.035751Z","iopub.status.idle":"2022-07-27T15:57:24.902740Z","shell.execute_reply.started":"2022-07-27T15:57:01.035720Z","shell.execute_reply":"2022-07-27T15:57:24.901585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df['review'][0])\ndf['review']=t.texts_to_sequences(df['review'])\nprint(df['review'][0])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:24.904911Z","iopub.execute_input":"2022-07-27T15:57:24.905367Z","iopub.status.idle":"2022-07-27T15:57:43.986275Z","shell.execute_reply.started":"2022-07-27T15:57:24.905325Z","shell.execute_reply":"2022-07-27T15:57:43.985143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train,x_test,y_train,y_test = train_test_split(df['review'],df['review_status'], test_size = 0.2)\n\nprint(\"x_train\")\nprint(x_train.shape)\n\nprint(\"x_test\")\nprint(x_test.shape)\n\nprint(\"y_train\")\nprint(y_train.shape)\n\nprint(\"y_test\")\nprint(y_test.shape)\n\nreview_length = [len(x) for x in x_train]\nprint(max(review_length))\nprint(min(review_length))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:43.987691Z","iopub.execute_input":"2022-07-27T15:57:43.988036Z","iopub.status.idle":"2022-07-27T15:57:44.364964Z","shell.execute_reply.started":"2022-07-27T15:57:43.988006Z","shell.execute_reply":"2022-07-27T15:57:44.363721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_limit = 1 # first 500 words only of each review to be considered\nx_train = sequence.pad_sequences(x_train,maxlen=input_limit) # pad with 0 if length is less than 500\nx_test = sequence.pad_sequences(x_test,maxlen=input_limit) # pad with 0 if length is less than 500","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:44.366185Z","iopub.execute_input":"2022-07-27T15:57:44.366959Z","iopub.status.idle":"2022-07-27T15:57:47.398256Z","shell.execute_reply.started":"2022-07-27T15:57:44.366923Z","shell.execute_reply":"2022-07-27T15:57:47.397081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:47.400425Z","iopub.execute_input":"2022-07-27T15:57:47.401161Z","iopub.status.idle":"2022-07-27T15:57:47.411465Z","shell.execute_reply.started":"2022-07-27T15:57:47.401122Z","shell.execute_reply":"2022-07-27T15:57:47.410610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ofcourse we need to do one hot encoding here! \nprint(y_train[0]) #before one hot encoding\nfrom keras.utils import np_utils\n#y_train = np_utils.to_categorical(y_train-1,num_classes=2)\n#y_test = np_utils.to_categorical(y_test-1,num_classes=2)\ny_train = np.asarray(y_train).astype('float32').reshape((-1,1))\ny_test = np.asarray(y_test).astype('float32').reshape((-1,1))\ny_train[0] # How label looks after one encoding","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:47.414445Z","iopub.execute_input":"2022-07-27T15:57:47.415065Z","iopub.status.idle":"2022-07-27T15:57:47.441857Z","shell.execute_reply.started":"2022-07-27T15:57:47.415033Z","shell.execute_reply":"2022-07-27T15:57:47.441065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test,x_valid,y_test,y_valid = train_test_split(x_test,y_test,test_size=0.5)\nprint(\"x_valid\")\nprint(x_train.shape)\n\nprint(\"x_test\")\nprint(x_test.shape)\n\nprint(\"y_valid\")\nprint(y_train.shape)\n\nprint(\"y_test\")\nprint(y_test.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:47.443067Z","iopub.execute_input":"2022-07-27T15:57:47.443849Z","iopub.status.idle":"2022-07-27T15:57:47.455304Z","shell.execute_reply.started":"2022-07-27T15:57:47.443812Z","shell.execute_reply":"2022-07-27T15:57:47.454530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. MLP","metadata":{}},{"cell_type":"code","source":"model = Sequential()\n# We choose to represnt each word as a 16 element vector=output_dim\n# Note that the output of Embedding layer will be MAX_WORDS*16 i.e. each of the MAX_WORDS will be\n# represented as a 16 element vector.  \nmodel.add(Embedding(input_dim=max_words, output_dim=16,input_length=input_limit))\nmodel.add(Flatten())\n#model.add(Dense(100,activation='relu'))\nmodel.add(Dense(100,activation='sigmoid'))\nmodel.add(Dense(1,activation='softmax'))\n#model.compile(optimizer='adam',metrics=['accuracy'],loss='categorical_crossentropy')\nmodel.compile(optimizer='adam',metrics=['accuracy'],loss='binary_crossentropy')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:47.456859Z","iopub.execute_input":"2022-07-27T15:57:47.458196Z","iopub.status.idle":"2022-07-27T15:57:47.613717Z","shell.execute_reply.started":"2022-07-27T15:57:47.458152Z","shell.execute_reply":"2022-07-27T15:57:47.612616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's fit the model\nfrom keras.callbacks import ModelCheckpoint\nmodel_checkpoint = ModelCheckpoint('best.hdf5',save_best_only=True)\nhistory = model.fit(x_train,y_train, validation_data=(x_valid,y_valid),batch_size=64, epochs=10, callbacks=[model_checkpoint])\n\nloss, accuracy = model.evaluate(x_train, y_train, verbose=False)\nprint(\"Training Accuracy: {:.4f}\".format(accuracy))\n\nloss, accuracy = model.evaluate(x_valid, y_valid, verbose=False)\nprint(\"Testing Accuracy:  {:.4f}\".format(accuracy))\n\nresult = model.predict(x_valid)\nprint(classification_report(y_valid.argmax(axis=1), result.argmax(axis=1)))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T15:57:47.615309Z","iopub.execute_input":"2022-07-27T15:57:47.615774Z","iopub.status.idle":"2022-07-27T16:02:37.714101Z","shell.execute_reply.started":"2022-07-27T15:57:47.615730Z","shell.execute_reply":"2022-07-27T16:02:37.712745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Time to check for accuracy\nmodel.load_weights('best.hdf5')\nscore= model.evaluate(x_test,y_test)\nscore[1]\nfinalmix_accuracy = score[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:02:37.715469Z","iopub.execute_input":"2022-07-27T16:02:37.715814Z","iopub.status.idle":"2022-07-27T16:02:41.366787Z","shell.execute_reply.started":"2022-07-27T16:02:37.715782Z","shell.execute_reply":"2022-07-27T16:02:41.365887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('ggplot')\n\ndef plot_history(history):\n    acc = history.history['accuracy']\n    val_acc = history.history['val_accuracy']\n    loss = history.history['loss']\n    val_loss = history.history['val_loss']\n    x = range(1, len(acc) + 1)\n\n    plt.figure(figsize=(12, 5))\n    plt.subplot(1, 2, 1)\n    plt.plot(x, acc, 'b', label='Training accuracy')\n    plt.plot(x, val_acc, 'r', label='Validation accuracy')\n    plt.title('Training and validation accuracy')\n    plt.legend()\n    plt.subplot(1, 2, 2)\n    plt.plot(x, loss, 'b', label='Training loss')\n    plt.plot(x, val_loss, 'r', label='Validation loss')\n    plt.title('Training and validation loss')\n    plt.legend()\n    \nplot_history(history)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:02:41.368616Z","iopub.execute_input":"2022-07-27T16:02:41.369073Z","iopub.status.idle":"2022-07-27T16:02:41.761261Z","shell.execute_reply.started":"2022-07-27T16:02:41.369029Z","shell.execute_reply":"2022-07-27T16:02:41.760091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Submit - Finalmix","metadata":{}},{"cell_type":"code","source":"y_pred = model.predict(x_valid, batch_size=1024)\nprint(y_pred[0])\n#submission = pd.read_csv('../input/../input/steam-user-review-20220725/reviews_clean_v2.csv',low_memory=False)\n#submission[[1,0]] = y_pred\n#submission.to_csv('submission.csv', index=False)\noutput = pd.DataFrame({\n\"Disease\": [finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy,finalmix_accuracy],\n\"Symptom_0\": [y_pred[0],y_pred[0],y_pred[0],y_pred[0],y_pred[0],y_pred[0],y_pred[0],y_pred[0],y_pred[0]]\n})\n\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-27T16:03:32.184962Z","iopub.execute_input":"2022-07-27T16:03:32.185927Z","iopub.status.idle":"2022-07-27T16:03:32.362438Z","shell.execute_reply.started":"2022-07-27T16:03:32.185875Z","shell.execute_reply":"2022-07-27T16:03:32.361235Z"},"trusted":true},"execution_count":null,"outputs":[]}]}