{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# import basic library","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport random\nimport warnings\nwarnings.simplefilter('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:51.371246Z","iopub.execute_input":"2022-07-24T11:57:51.371591Z","iopub.status.idle":"2022-07-24T11:57:54.558795Z","shell.execute_reply.started":"2022-07-24T11:57:51.371560Z","shell.execute_reply":"2022-07-24T11:57:54.557667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# import automl library","metadata":{}},{"cell_type":"code","source":"from IPython.display import clear_output\n!pip3 install flaml \nclear_output()\n\nfrom flaml import AutoML","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# define variables","metadata":{}},{"cell_type":"code","source":"TRAIN_PATH = \"../input/house-prices-advanced-regression-techniques/train.csv\"\nTEST_PATH = \"../input/house-prices-advanced-regression-techniques/test.csv\"\nSAMPLE_SUBMISSION_PATH = \"../input/house-prices-advanced-regression-techniques/sample_submission.csv\"\nSUBMISSION_PATH = \"submission.csv\"\n\nID = \"Id\"\nTARGET = \"SalePrice\"\n\nSEED = 1004\ndef seed_everything(seed=SEED):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n\nseed_everything()\n\nMODEL_TIME_BUDGET = 60*3\nMODEL_METRIC = 'rmse'\nMODEL_TASK = \"regression\"\nMODEL_LIST = ['lgbm']\nMODEL_LOG_FILE_PATH = \"flaml_log.log\"\n\nCROSS_VAL_CV = 3","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:54.560147Z","iopub.execute_input":"2022-07-24T11:57:54.560452Z","iopub.status.idle":"2022-07-24T11:57:54.568141Z","shell.execute_reply.started":"2022-07-24T11:57:54.560424Z","shell.execute_reply":"2022-07-24T11:57:54.567273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# null check and fill new data ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_PATH)\ntest = pd.read_csv(TEST_PATH)\n\ndef checkNull_fillData(df):\n    for col in df.columns:\n        if len(df.loc[df[col].isnull() == True]) != 0:\n            if df[col].dtype == \"float64\" or df[col].dtype == \"int64\":\n                df.loc[df[col].isnull() == True,col] = df[col].median()\n            else:\n                df.loc[df[col].isnull() == True,col] = \"Missing\"\n                \ncheckNull_fillData(train)\ncheckNull_fillData(test)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:54.570753Z","iopub.execute_input":"2022-07-24T11:57:54.571369Z","iopub.status.idle":"2022-07-24T11:57:54.802045Z","shell.execute_reply.started":"2022-07-24T11:57:54.571335Z","shell.execute_reply":"2022-07-24T11:57:54.800652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Label Encoding","metadata":{}},{"cell_type":"code","source":"col_names = []\nfor col in train:\n    if train[col].dtypes == \"object\":\n        col_names.append(col)\n        \nfrom sklearn.preprocessing import LabelEncoder\n\n\n\nfor col in col_names:\n    encoder = LabelEncoder()\n    encoder.fit(train[col])\n    train[col] = encoder.transform(train[col])\n\n    for label in np.unique(test[col]):\n        if label not in encoder.classes_: \n            encoder.classes_ = np.append(encoder.classes_, label) \n    test[col] = encoder.transform(test[col])","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:54.803919Z","iopub.execute_input":"2022-07-24T11:57:54.804544Z","iopub.status.idle":"2022-07-24T11:57:54.933746Z","shell.execute_reply.started":"2022-07-24T11:57:54.804497Z","shell.execute_reply":"2022-07-24T11:57:54.932930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# feature engineering - make mean column","metadata":{}},{"cell_type":"code","source":"GROUP_FUNCTION = \"mean\"\nENCODED = \"_encoded\"\n\nfor col in col_names: \n    train[col + ENCODED] = train.groupby(col)[TARGET].transform(GROUP_FUNCTION)\n    di = train[[col,col + ENCODED]].drop_duplicates().set_index(col).to_dict()[col + ENCODED]\n    train= train.replace({col:di})\n    test= test.replace({col:di})\n    \n    train = train.drop([col + ENCODED],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:54.935062Z","iopub.execute_input":"2022-07-24T11:57:54.935633Z","iopub.status.idle":"2022-07-24T11:57:55.334072Z","shell.execute_reply.started":"2022-07-24T11:57:54.935598Z","shell.execute_reply":"2022-07-24T11:57:55.332951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build FLAML Model","metadata":{}},{"cell_type":"code","source":"X = train.drop([ID,TARGET],axis=1)\ny = train[TARGET]\n\nmodel = AutoML()\nparams = {\n    \"time_budget\": MODEL_TIME_BUDGET,  \n    \"metric\": MODEL_METRIC,\n    \"estimator_list\": MODEL_LIST, \n    \"task\": MODEL_TASK,\n    \"seed\":SEED,\n    \"log_file_name\":MODEL_LOG_FILE_PATH,\n}\nmodel.fit(X, y, **params)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T11:57:55.335472Z","iopub.execute_input":"2022-07-24T11:57:55.335809Z","iopub.status.idle":"2022-07-24T12:02:55.328451Z","shell.execute_reply.started":"2022-07-24T11:57:55.335779Z","shell.execute_reply":"2022-07-24T12:02:55.327421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check Model","metadata":{}},{"cell_type":"code","source":"model.model.model","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:02:55.330018Z","iopub.execute_input":"2022-07-24T12:02:55.330358Z","iopub.status.idle":"2022-07-24T12:02:55.344584Z","shell.execute_reply.started":"2022-07-24T12:02:55.330327Z","shell.execute_reply":"2022-07-24T12:02:55.343093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate Model","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nneg_score = cross_val_score(model.model.model, X, y, cv=CROSS_VAL_CV,scoring='neg_mean_squared_error')\nscore = np.sqrt(-neg_score)\nscore","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:02:55.346225Z","iopub.execute_input":"2022-07-24T12:02:55.346600Z","iopub.status.idle":"2022-07-24T12:02:57.634989Z","shell.execute_reply.started":"2022-07-24T12:02:55.346569Z","shell.execute_reply":"2022-07-24T12:02:57.633915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict Test Data","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv(SAMPLE_SUBMISSION_PATH)\nsub[TARGET] = model.predict(test)\nsub.to_csv(SUBMISSION_PATH,index=False)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-24T12:02:57.636788Z","iopub.execute_input":"2022-07-24T12:02:57.637260Z","iopub.status.idle":"2022-07-24T12:02:57.703134Z","shell.execute_reply.started":"2022-07-24T12:02:57.637217Z","shell.execute_reply":"2022-07-24T12:02:57.701904Z"},"trusted":true},"execution_count":null,"outputs":[]}]}