{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T09:24:42.801115Z","iopub.execute_input":"2022-07-13T09:24:42.801702Z","iopub.status.idle":"2022-07-13T09:24:42.832433Z","shell.execute_reply.started":"2022-07-13T09:24:42.801572Z","shell.execute_reply":"2022-07-13T09:24:42.831283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/goodreads-books-reviews-290312/goodreads_train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:24:42.835164Z","iopub.execute_input":"2022-07-13T09:24:42.835819Z","iopub.status.idle":"2022-07-13T09:25:04.537015Z","shell.execute_reply.started":"2022-07-13T09:24:42.835782Z","shell.execute_reply":"2022-07-13T09:25:04.535925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Decreasing memory usage to avoid memory allocation error.","metadata":{}},{"cell_type":"code","source":"def reduce_mem_usage(train_data):\n    start_mem_original = train_data.memory_usage().sum()\n    print('Original Memory usage of dataframe is {:.2f} MB'.format(start_mem_original))\n    start_mem = train_data.memory_usage().sum() / 1024**2\n    print('Updated Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in train_data.columns:\n        col_type = train_data[col].dtype\n\n    if col_type != object:\n        c_min = train_data[col].min()\n        c_max = train_data[col].max()\n        if str(col_type)[:3] == 'int':\n            if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                train_data[col] = train_data[col].astype(np.int8)\n            elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                train_data[col] = train_data[col].astype(np.int16)\n            elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                train_data[col] = train_data[col].astype(np.int32)\n            elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                train_data[col] = train_data[col].astype(np.int64)  \n        else:\n            if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                train_data[col] = train_data[col].astype(np.float16)\n            elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                train_data[col] = train_data[col].astype(np.float32)\n            else:\n                train_data[col] = train_data[col].astype(np.float64)\n    else:\n        train_data[col] = train_data[col].astype('category')\n\n    end_mem = train_data.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n\n    return train_data","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:26:15.151803Z","iopub.execute_input":"2022-07-13T09:26:15.152479Z","iopub.status.idle":"2022-07-13T09:26:15.169167Z","shell.execute_reply.started":"2022-07-13T09:26:15.152442Z","shell.execute_reply":"2022-07-13T09:26:15.167906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reduce_mem_usage(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:21.734331Z","iopub.execute_input":"2022-07-13T09:28:21.734832Z","iopub.status.idle":"2022-07-13T09:28:21.775254Z","shell.execute_reply.started":"2022-07-13T09:28:21.734798Z","shell.execute_reply":"2022-07-13T09:28:21.773327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Dropping unneccesary variables","metadata":{}},{"cell_type":"code","source":"train = train.drop(columns=['user_id', 'book_id' ,'date_added', 'date_updated', 'read_at', 'started_at'], axis=0)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:27.909358Z","iopub.execute_input":"2022-07-13T09:28:27.910130Z","iopub.status.idle":"2022-07-13T09:28:27.969160Z","shell.execute_reply.started":"2022-07-13T09:28:27.910094Z","shell.execute_reply":"2022-07-13T09:28:27.968102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:29.661754Z","iopub.execute_input":"2022-07-13T09:28:29.662423Z","iopub.status.idle":"2022-07-13T09:28:29.672214Z","shell.execute_reply.started":"2022-07-13T09:28:29.662378Z","shell.execute_reply":"2022-07-13T09:28:29.671028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = train['rating']\nX = train.drop('rating', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:29.837990Z","iopub.execute_input":"2022-07-13T09:28:29.838310Z","iopub.status.idle":"2022-07-13T09:28:29.889548Z","shell.execute_reply.started":"2022-07-13T09:28:29.838283Z","shell.execute_reply":"2022-07-13T09:28:29.888648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:00:12.280044Z","iopub.execute_input":"2022-07-12T11:00:12.281265Z","iopub.status.idle":"2022-07-12T11:00:12.293294Z","shell.execute_reply.started":"2022-07-12T11:00:12.281205Z","shell.execute_reply":"2022-07-12T11:00:12.292032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:00:12.883091Z","iopub.execute_input":"2022-07-12T11:00:12.883483Z","iopub.status.idle":"2022-07-12T11:00:12.890791Z","shell.execute_reply.started":"2022-07-12T11:00:12.883452Z","shell.execute_reply":"2022-07-12T11:00:12.889755Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ### Review id - One hot encoding","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:33.458629Z","iopub.execute_input":"2022-07-13T09:28:33.459272Z","iopub.status.idle":"2022-07-13T09:28:33.883962Z","shell.execute_reply.started":"2022-07-13T09:28:33.459237Z","shell.execute_reply":"2022-07-13T09:28:33.882983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['review_id'] = le.fit_transform(X['review_id'])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:33.885801Z","iopub.execute_input":"2022-07-13T09:28:33.886139Z","iopub.status.idle":"2022-07-13T09:28:37.286827Z","shell.execute_reply.started":"2022-07-13T09:28:33.886112Z","shell.execute_reply":"2022-07-13T09:28:37.285835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:00:19.234256Z","iopub.execute_input":"2022-07-12T11:00:19.235000Z","iopub.status.idle":"2022-07-12T11:00:19.245953Z","shell.execute_reply.started":"2022-07-12T11:00:19.234962Z","shell.execute_reply":"2022-07-12T11:00:19.244939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Review text - tokenization","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nimport tensorflow as tf\nimport random","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:28:37.288854Z","iopub.execute_input":"2022-07-13T09:28:37.289226Z","iopub.status.idle":"2022-07-13T09:28:42.829263Z","shell.execute_reply.started":"2022-07-13T09:28:37.289191Z","shell.execute_reply":"2022-07-13T09:28:42.828029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = Tokenizer(num_words=10000, oov_token='<UNK>')\n\ntokenizer.fit_on_texts(X['review_text'])\n\n#print(tokenizer.texts_to_sequences(X['review_text'][0]))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:31:35.924292Z","iopub.execute_input":"2022-07-13T09:31:35.925260Z","iopub.status.idle":"2022-07-13T09:33:39.693206Z","shell.execute_reply.started":"2022-07-13T09:31:35.925225Z","shell.execute_reply":"2022-07-13T09:33:39.692203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(tokenizer.word_index))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:33:39.695227Z","iopub.execute_input":"2022-07-13T09:33:39.695626Z","iopub.status.idle":"2022-07-13T09:33:39.701214Z","shell.execute_reply.started":"2022-07-13T09:33:39.695577Z","shell.execute_reply":"2022-07-13T09:33:39.699997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:34:09.625848Z","iopub.execute_input":"2022-07-13T09:34:09.626200Z","iopub.status.idle":"2022-07-13T09:34:09.631174Z","shell.execute_reply.started":"2022-07-13T09:34:09.626172Z","shell.execute_reply":"2022-07-13T09:34:09.630166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lengths = [len(t.split(' ')) for t in X['review_text']]\n\nplt.hist(lengths, bins=len(set(lengths)))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:34:09.792156Z","iopub.execute_input":"2022-07-13T09:34:09.792817Z","iopub.status.idle":"2022-07-13T09:34:25.941519Z","shell.execute_reply.started":"2022-07-13T09:34:09.792777Z","shell.execute_reply":"2022-07-13T09:34:25.939658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(max(lengths))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:34:25.943425Z","iopub.execute_input":"2022-07-13T09:34:25.943903Z","iopub.status.idle":"2022-07-13T09:34:25.964900Z","shell.execute_reply.started":"2022-07-13T09:34:25.943849Z","shell.execute_reply":"2022-07-13T09:34:25.963848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_sequences(tokenizer, review):\n    sequences = tokenizer.texts_to_sequences(review)\n    padded_sequences = pad_sequences(sequences, truncating='post', maxlen=1500, padding='post')\n    return padded_sequences","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:34:35.411881Z","iopub.execute_input":"2022-07-13T09:34:35.412584Z","iopub.status.idle":"2022-07-13T09:34:35.417949Z","shell.execute_reply.started":"2022-07-13T09:34:35.412546Z","shell.execute_reply":"2022-07-13T09:34:35.416928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X['review_text'] = get_sequences(tokenizer, X['review_text'])\nX['review_text'][0]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T09:34:38.867427Z","iopub.execute_input":"2022-07-13T09:34:38.868143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Despite of decreasing the data usage, it still shows \"your notebook tried to allocate more memory than is available. It has restarted\" after above line. Any suggestions?","metadata":{}},{"cell_type":"markdown","source":"> ## Data Modelling","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.Sequential([\n    tf.keras.layers.Embedding(50000, 16, input_length=35),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(20, return_sequences=True)),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(20)),\n    tf.keras.layers.Dense(6, activation='softmax')\n])\n\nmodel.compile(\n    loss='sparse_categorical_crossentropy',\n    optimizer='adam',\n    metrics=['accuracy']\n)\n\nmodel.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"h = model.fit(\n    X, y,\n    epochs=2,\n    callbacks=[\n        tf.keras.callbacks.EarlyStopping(monitor='accuracy', patience=2)\n    ]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\nrfr = RandomForestRegressor()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T08:08:32.575223Z","iopub.execute_input":"2022-07-12T08:08:32.575518Z","iopub.status.idle":"2022-07-12T08:08:32.793957Z","shell.execute_reply.started":"2022-07-12T08:08:32.575491Z","shell.execute_reply":"2022-07-12T08:08:32.792838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfr.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T11:05:57.925942Z","iopub.execute_input":"2022-07-12T11:05:57.926440Z","iopub.status.idle":"2022-07-12T11:05:57.944658Z","shell.execute_reply.started":"2022-07-12T11:05:57.926335Z","shell.execute_reply":"2022-07-12T11:05:57.943503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ## Thank you!","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}