{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h3>How Good Is TPOT For Advanced House Prediction competition? Let's see!</h3>","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T19:08:59.026512Z","iopub.execute_input":"2022-08-12T19:08:59.027067Z","iopub.status.idle":"2022-08-12T19:08:59.061239Z","shell.execute_reply.started":"2022-08-12T19:08:59.026964Z","shell.execute_reply":"2022-08-12T19:08:59.060194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:08:59.063502Z","iopub.execute_input":"2022-08-12T19:08:59.064814Z","iopub.status.idle":"2022-08-12T19:09:00.151459Z","shell.execute_reply.started":"2022-08-12T19:08:59.064749Z","shell.execute_reply":"2022-08-12T19:09:00.150210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading our data...**","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/train.csv').set_index('Id')\ntest_df = pd.read_csv('/kaggle/input/house-prices-advanced-regression-techniques/test.csv').set_index('Id')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.153611Z","iopub.execute_input":"2022-08-12T19:09:00.154033Z","iopub.status.idle":"2022-08-12T19:09:00.245209Z","shell.execute_reply.started":"2022-08-12T19:09:00.153995Z","shell.execute_reply":"2022-08-12T19:09:00.243615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n\n\n\n\n**Checking basic info...**","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.246650Z","iopub.execute_input":"2022-08-12T19:09:00.247027Z","iopub.status.idle":"2022-08-12T19:09:00.287863Z","shell.execute_reply.started":"2022-08-12T19:09:00.246992Z","shell.execute_reply":"2022-08-12T19:09:00.286689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.options.display.min_rows = 115\nprint(train_df.isnull().sum().sort_values(ascending=False))\nprint(test_df.isnull().sum().sort_values(ascending=False))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.290649Z","iopub.execute_input":"2022-08-12T19:09:00.291735Z","iopub.status.idle":"2022-08-12T19:09:00.333184Z","shell.execute_reply.started":"2022-08-12T19:09:00.291694Z","shell.execute_reply":"2022-08-12T19:09:00.332217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Let's impute missing data</h3>","metadata":{}},{"cell_type":"code","source":"list_null_features = ['PoolQC', 'MiscFeature', 'Alley', 'Fence', 'FireplaceQu', 'LotFrontage', 'GarageYrBlt', 'GarageCond',\n                     'GarageType', 'GarageFinish', 'GarageQual', 'BsmtFinType1', 'MasVnrArea', 'MasVnrType', 'Electrical']\n\nlist_to_replace = list(train_df[list_null_features].select_dtypes(exclude='object').columns)\n\nprint(list_to_replace)\n\nfor i in list_to_replace:\n    train_df[i].fillna((train_df[i].mean()), inplace=True)\n    test_df[i].fillna((train_df[i].mean()), inplace=True)\n    print(f'Replacing {i}')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.334643Z","iopub.execute_input":"2022-08-12T19:09:00.336132Z","iopub.status.idle":"2022-08-12T19:09:00.358343Z","shell.execute_reply.started":"2022-08-12T19:09:00.336081Z","shell.execute_reply":"2022-08-12T19:09:00.357133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Ordinal Encoder</h3>","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\n\nobject_columns = list(train_df.select_dtypes(include=['object']).columns)\n\nord_encoder = OrdinalEncoder()\n\nfor column in object_columns:\n    train_df[column] = ord_encoder.fit_transform(train_df[[column]])\n    test_df[column] = ord_encoder.fit_transform(test_df[[column]])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.359892Z","iopub.execute_input":"2022-08-12T19:09:00.360264Z","iopub.status.idle":"2022-08-12T19:09:00.654686Z","shell.execute_reply.started":"2022-08-12T19:09:00.360220Z","shell.execute_reply":"2022-08-12T19:09:00.653019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.fillna(0)\ntest_df = test_df.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.656373Z","iopub.execute_input":"2022-08-12T19:09:00.656741Z","iopub.status.idle":"2022-08-12T19:09:00.665734Z","shell.execute_reply.started":"2022-08-12T19:09:00.656709Z","shell.execute_reply":"2022-08-12T19:09:00.664479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Spliting our data</h3>","metadata":{}},{"cell_type":"code","source":"X = train_df.copy()\ny = X.pop('SalePrice')\n\nfrom sklearn.model_selection import cross_val_score, train_test_split, GridSearchCV\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=23)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.667343Z","iopub.execute_input":"2022-08-12T19:09:00.668130Z","iopub.status.idle":"2022-08-12T19:09:00.757274Z","shell.execute_reply.started":"2022-08-12T19:09:00.668078Z","shell.execute_reply":"2022-08-12T19:09:00.756074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Let's set an earlier stop of 7. If in 7 generation the model doesn't improve, TPOT will stop running. This will take a lot of time to run!</h3>","metadata":{}},{"cell_type":"code","source":"#from tpot import TPOTRegressor\n\n#pipeline_optimizer  = TPOTRegressor(generations=100, population_size=100,\n #                        offspring_size=None, mutation_rate=0.9,\n  #                       crossover_rate=0.1,\n   #                      scoring='neg_mean_squared_error', cv=5,\n    #                     subsample=1.0, n_jobs=-1,\n     #                    max_time_mins=None, max_eval_time_mins=5,\n      #                   random_state=None, config_dict=None,\n       #                  template=None,\n        #                 warm_start=True,\n         #                memory=None,\n          #               use_dask=False,\n           #              periodic_checkpoint_folder=None,\n            #             early_stop=7,\n             #            verbosity=2,\n              #           disable_update_check=False)\n\n\n#pipeline_optimizer.fit(X_train, y_train)\n#print(pipeline_optimizer.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.759227Z","iopub.execute_input":"2022-08-12T19:09:00.760087Z","iopub.status.idle":"2022-08-12T19:09:00.765613Z","shell.execute_reply.started":"2022-08-12T19:09:00.760041Z","shell.execute_reply":"2022-08-12T19:09:00.764382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Exporting the file to see the optimized pipeline that TPOT made for us</h3>","metadata":{}},{"cell_type":"code","source":"#pipeline_optimizer.export('tpot.txt')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.769671Z","iopub.execute_input":"2022-08-12T19:09:00.770102Z","iopub.status.idle":"2022-08-12T19:09:00.778491Z","shell.execute_reply.started":"2022-08-12T19:09:00.770056Z","shell.execute_reply":"2022-08-12T19:09:00.777487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Here is the TPOT Optimized pipeline for our dataset. Let's run this model and submit to see how it goes.</h3>","metadata":{}},{"cell_type":"code","source":"#This code is already \"cleaned\". TPOT export is a little bit different \n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.linear_model import ElasticNetCV\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import make_pipeline, make_union\nfrom sklearn.svm import LinearSVR\nfrom tpot.builtins import StackingEstimator\nfrom xgboost import XGBRegressor\n\n\n# Average CV score on the training set was: -689066004.1415527\nexported_pipeline = make_pipeline(\n    StackingEstimator(estimator=LinearSVR(C=0.1, dual=True, epsilon=0.01, loss=\"squared_epsilon_insensitive\", tol=1e-05)),\n    StackingEstimator(estimator=ElasticNetCV(l1_ratio=0.6000000000000001, tol=0.1)),\n    XGBRegressor(learning_rate=0.1, max_depth=6, min_child_weight=2, n_estimators=100, n_jobs=1, objective=\"reg:squarederror\", subsample=0.9000000000000001, verbosity=0)\n)\n\nexported_pipeline.fit(X, y)\nresults = exported_pipeline.predict(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:00.781641Z","iopub.execute_input":"2022-08-12T19:09:00.782239Z","iopub.status.idle":"2022-08-12T19:09:05.373746Z","shell.execute_reply.started":"2022-08-12T19:09:00.782189Z","shell.execute_reply":"2022-08-12T19:09:05.372094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>Exporting the submission!!</h3>","metadata":{}},{"cell_type":"code","source":"test_df = test_df.rename(columns={\"target\":\"SalePrice\"})\n\nFinal = pd.DataFrame(columns=[\"Id\",\"SalePrice\"])\nFinal[\"Id\"] = test_df.index\nFinal[\"SalePrice\"] = results\nFinal[\"Id\"] = Final[\"Id\"].astype(int)\nFinal.set_index('Id', inplace=True)\nFinal.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T19:09:05.376221Z","iopub.execute_input":"2022-08-12T19:09:05.377631Z","iopub.status.idle":"2022-08-12T19:09:05.420846Z","shell.execute_reply.started":"2022-08-12T19:09:05.377562Z","shell.execute_reply":"2022-08-12T19:09:05.419116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3>We scored 0.12741 (834 position) at House Prices - Advanced Regression Techniques competition! It is a great start!</h3>","metadata":{}}]}