{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.activations import linear, relu, sigmoid\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\n%matplotlib widget\nimport matplotlib.pyplot as plt\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T13:44:13.810958Z","iopub.execute_input":"2022-08-11T13:44:13.811335Z","iopub.status.idle":"2022-08-11T13:44:13.824404Z","shell.execute_reply.started":"2022-08-11T13:44:13.811303Z","shell.execute_reply":"2022-08-11T13:44:13.823613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preparation","metadata":{}},{"cell_type":"markdown","source":"After a quick analysis over the data, I consider all the features significant enough","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\ntest_df = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:54:04.713303Z","iopub.execute_input":"2022-08-11T13:54:04.713730Z","iopub.status.idle":"2022-08-11T13:54:04.762544Z","shell.execute_reply.started":"2022-08-11T13:54:04.713695Z","shell.execute_reply":"2022-08-11T13:54:04.761401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_df.iloc[:1168, :].SalePrice\ny_cv = train_df.iloc[1168:, :].SalePrice","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:54:05.534829Z","iopub.execute_input":"2022-08-11T13:54:05.535267Z","iopub.status.idle":"2022-08-11T13:54:05.541537Z","shell.execute_reply.started":"2022-08-11T13:54:05.535222Z","shell.execute_reply":"2022-08-11T13:54:05.540416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Drop some columns:","metadata":{}},{"cell_type":"code","source":"train_df = train_df.drop(['Id', 'SalePrice'], axis=1)\ntest_df = test_df.drop(['Id'], axis=1)\n\nlow_cardinality_cols = [cname for cname in train_df.columns if \n                                train_df[cname].nunique() < 9 and\n                                train_df[cname].dtype == \"object\"]\nnumeric_cols = [cname for cname in train_df.columns if \n                                train_df[cname].dtype in ['int64', 'float64']]\nmy_cols = low_cardinality_cols + numeric_cols\n\ntrain_df = train_df[my_cols]\ntest_df = test_df[my_cols]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:54:07.311863Z","iopub.execute_input":"2022-08-11T13:54:07.312291Z","iopub.status.idle":"2022-08-11T13:54:07.345434Z","shell.execute_reply.started":"2022-08-11T13:54:07.312254Z","shell.execute_reply":"2022-08-11T13:54:07.344423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our data set consists categorical data, which will not be accepted by our algorithm. That is why we one-hot encode all the values. The training set has a different number of categorials vs the test set. This is why we need to align both sets.","metadata":{}},{"cell_type":"code","source":"train_dummy = pd.get_dummies(train_df)\ntest_dummy = pd.get_dummies(test_df)\ntrain_dummy, test_dummy = train_dummy.align(test_dummy, join='left', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:54:08.445976Z","iopub.execute_input":"2022-08-11T13:54:08.446668Z","iopub.status.idle":"2022-08-11T13:54:08.525269Z","shell.execute_reply.started":"2022-08-11T13:54:08.446632Z","shell.execute_reply":"2022-08-11T13:54:08.524285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Since we do not want to lose a whole column if data is missing, I am going to fill it with the mean value instead","metadata":{}},{"cell_type":"code","source":"my_imputer = SimpleImputer()\ntrain_imp = my_imputer.fit_transform(train_dummy)\ntest_imp = my_imputer.transform(test_dummy)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:54:54.447733Z","iopub.execute_input":"2022-08-11T13:54:54.448130Z","iopub.status.idle":"2022-08-11T13:54:54.478098Z","shell.execute_reply.started":"2022-08-11T13:54:54.448100Z","shell.execute_reply":"2022-08-11T13:54:54.477129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After one-hot encoding we end up with:","metadata":{}},{"cell_type":"code","source":"scaler = StandardScaler()\ntrain_imp = scaler.fit_transform(train_imp)\ntest_imp = scaler.transform(test_imp)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:55:02.214225Z","iopub.execute_input":"2022-08-11T13:55:02.214649Z","iopub.status.idle":"2022-08-11T13:55:02.227127Z","shell.execute_reply.started":"2022-08-11T13:55:02.214612Z","shell.execute_reply":"2022-08-11T13:55:02.226072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Seperate the training set into\n* train set\n* cv set","metadata":{}},{"cell_type":"code","source":"df_train = train_imp[:1168, :] # 80% of training data set\ndf_cv = train_imp[1168:, :] # 20% of training data set","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:44:31.047680Z","iopub.execute_input":"2022-08-11T13:44:31.048545Z","iopub.status.idle":"2022-08-11T13:44:31.054771Z","shell.execute_reply.started":"2022-08-11T13:44:31.048505Z","shell.execute_reply":"2022-08-11T13:44:31.053145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"markdown","source":"In order to know whether the model underfits, we'd like to know the mean value of the price","metadata":{}},{"cell_type":"code","source":"look_mean = pd.read_csv('../input/house-prices-advanced-regression-techniques/train.csv')\nmean = look_mean[\"SalePrice\"].mean()\nmean","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:44:33.170034Z","iopub.execute_input":"2022-08-11T13:44:33.170755Z","iopub.status.idle":"2022-08-11T13:44:33.201287Z","shell.execute_reply.started":"2022-08-11T13:44:33.170704Z","shell.execute_reply":"2022-08-11T13:44:33.200049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The mean price is **180921**\n\n10% of 180 000 is **18 000**, which squared is exactly **3 240 000**. We are looking for a configuration that has **J error < 3 240 000**","metadata":{}},{"cell_type":"code","source":"model = Sequential(\n    [               \n        tf.keras.layers.Dense(200, activation='relu'),        \n        tf.keras.layers.Dense(100, activation='relu'),\n        tf.keras.layers.Dense(50, activation='relu'),\n        tf.keras.layers.Dense(25, activation='relu'),\n        tf.keras.layers.Dense(12, activation='relu'),\n        tf.keras.layers.Dense(1, activation='linear'),\n    ], name = \"my_model\" \n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:44:35.163755Z","iopub.execute_input":"2022-08-11T13:44:35.164428Z","iopub.status.idle":"2022-08-11T13:44:35.178475Z","shell.execute_reply.started":"2022-08-11T13:44:35.164391Z","shell.execute_reply":"2022-08-11T13:44:35.177457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(\n    loss=tf.keras.losses.MeanSquaredError(),\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.1),\n)\n\nhistory = model.fit(\n    df_train,y_train,\n    epochs=1000\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:47:13.302782Z","iopub.execute_input":"2022-08-11T13:47:13.303183Z","iopub.status.idle":"2022-08-11T13:47:24.112069Z","shell.execute_reply.started":"2022-08-11T13:47:13.303151Z","shell.execute_reply":"2022-08-11T13:47:24.110933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(df_cv, y_cv)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:11:43.613162Z","iopub.execute_input":"2022-08-11T14:11:43.613605Z","iopub.status.idle":"2022-08-11T14:11:43.715226Z","shell.execute_reply.started":"2022-08-11T14:11:43.613569Z","shell.execute_reply":"2022-08-11T14:11:43.713856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare for evaluation","metadata":{}},{"cell_type":"code","source":"predictions = model.predict(test_imp)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:55:10.240966Z","iopub.execute_input":"2022-08-11T13:55:10.241835Z","iopub.status.idle":"2022-08-11T13:55:10.494640Z","shell.execute_reply.started":"2022-08-11T13:55:10.241787Z","shell.execute_reply":"2022-08-11T13:55:10.493758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testtest = pd.read_csv('../input/house-prices-advanced-regression-techniques/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T13:56:20.871827Z","iopub.execute_input":"2022-08-11T13:56:20.872233Z","iopub.status.idle":"2022-08-11T13:56:20.904156Z","shell.execute_reply.started":"2022-08-11T13:56:20.872185Z","shell.execute_reply":"2022-08-11T13:56:20.902877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = tf.reshape(predictions, 1459).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:00:34.251735Z","iopub.execute_input":"2022-08-11T14:00:34.252141Z","iopub.status.idle":"2022-08-11T14:00:34.258665Z","shell.execute_reply.started":"2022-08-11T14:00:34.252107Z","shell.execute_reply":"2022-08-11T14:00:34.257460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'Id': testtest.Id, 'SalePrice': predictions})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T14:12:11.958983Z","iopub.execute_input":"2022-08-11T14:12:11.959515Z","iopub.status.idle":"2022-08-11T14:12:11.972538Z","shell.execute_reply.started":"2022-08-11T14:12:11.959466Z","shell.execute_reply":"2022-08-11T14:12:11.971276Z"},"trusted":true},"execution_count":null,"outputs":[]}]}