{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T08:58:40.876681Z","iopub.execute_input":"2022-08-12T08:58:40.877088Z","iopub.status.idle":"2022-08-12T08:58:40.886761Z","shell.execute_reply.started":"2022-08-12T08:58:40.877055Z","shell.execute_reply":"2022-08-12T08:58:40.885816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:40.888795Z","iopub.execute_input":"2022-08-12T08:58:40.889566Z","iopub.status.idle":"2022-08-12T08:58:40.897048Z","shell.execute_reply.started":"2022-08-12T08:58:40.889526Z","shell.execute_reply":"2022-08-12T08:58:40.895967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/tabular-playground-series-aug-2021/train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:40.901252Z","iopub.execute_input":"2022-08-12T08:58:40.901582Z","iopub.status.idle":"2022-08-12T08:58:46.845116Z","shell.execute_reply.started":"2022-08-12T08:58:40.901556Z","shell.execute_reply":"2022-08-12T08:58:46.844033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:46.847036Z","iopub.execute_input":"2022-08-12T08:58:46.847640Z","iopub.status.idle":"2022-08-12T08:58:46.909664Z","shell.execute_reply.started":"2022-08-12T08:58:46.847602Z","shell.execute_reply":"2022-08-12T08:58:46.908628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_len = len(train)\naverage_values = train.mean(numeric_only=True)\ntrain = train.fillna(average_values)\nprint(f'training exemples : {len(train)},  {(1-len(train)/total_len)*100:.1f} % data not NaN')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:46.911331Z","iopub.execute_input":"2022-08-12T08:58:46.911669Z","iopub.status.idle":"2022-08-12T08:58:47.138170Z","shell.execute_reply.started":"2022-08-12T08:58:46.911642Z","shell.execute_reply":"2022-08-12T08:58:47.137131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_mat = train.corr()\nplt.figure(figsize=(20,20), dpi=100)\nsns.heatmap(corr_mat,annot=False, fmt=\".2f\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:47.141133Z","iopub.execute_input":"2022-08-12T08:58:47.141699Z","iopub.status.idle":"2022-08-12T08:58:55.668426Z","shell.execute_reply.started":"2022-08-12T08:58:47.141655Z","shell.execute_reply":"2022-08-12T08:58:55.667326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(train[\"loss\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:55.670196Z","iopub.execute_input":"2022-08-12T08:58:55.670826Z","iopub.status.idle":"2022-08-12T08:58:56.948980Z","shell.execute_reply.started":"2022-08-12T08:58:55.670785Z","shell.execute_reply":"2022-08-12T08:58:56.947883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train.drop([\"id\",\"loss\"],axis=1)\ny = train[\"loss\"]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:56.950382Z","iopub.execute_input":"2022-08-12T08:58:56.950783Z","iopub.status.idle":"2022-08-12T08:58:57.017121Z","shell.execute_reply.started":"2022-08-12T08:58:56.950754Z","shell.execute_reply":"2022-08-12T08:58:57.015902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install flaml","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:58:57.018447Z","iopub.execute_input":"2022-08-12T08:58:57.018861Z","iopub.status.idle":"2022-08-12T08:59:08.945460Z","shell.execute_reply.started":"2022-08-12T08:58:57.018827Z","shell.execute_reply":"2022-08-12T08:59:08.944033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from flaml import AutoML\nautoml = AutoML()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:08.947475Z","iopub.execute_input":"2022-08-12T08:59:08.947888Z","iopub.status.idle":"2022-08-12T08:59:11.314096Z","shell.execute_reply.started":"2022-08-12T08:59:08.947852Z","shell.execute_reply":"2022-08-12T08:59:11.313132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"automl.fit(X, y,task=\"regression\",metric='rmse',time_budget=60*5*1)# 5mins","metadata":{"execution":{"iopub.status.busy":"2022-08-12T08:59:11.315655Z","iopub.execute_input":"2022-08-12T08:59:11.316173Z","iopub.status.idle":"2022-08-12T09:00:29.645998Z","shell.execute_reply.started":"2022-08-12T08:59:11.316127Z","shell.execute_reply":"2022-08-12T09:00:29.644805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[](http://)","metadata":{}},{"cell_type":"code","source":"print('Best ML leaner:', automl.best_estimator)\nprint('Best hyperparmeter config:', automl.best_config)\nprint('Best rmse on validation data: {0:.4g}'.format(automl.best_loss))\nprint('Training duration of best run: {0:.4g} s'.format(automl.best_config_train_time))","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:00:29.649527Z","iopub.execute_input":"2022-08-12T09:00:29.649960Z","iopub.status.idle":"2022-08-12T09:00:29.656561Z","shell.execute_reply.started":"2022-08-12T09:00:29.649927Z","shell.execute_reply":"2022-08-12T09:00:29.655365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"../input/tabular-playground-series-aug-2021/test.csv\")\nprint(f'len : {len(test)}')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:00:29.658167Z","iopub.execute_input":"2022-08-12T09:00:29.658595Z","iopub.status.idle":"2022-08-12T09:00:33.079159Z","shell.execute_reply.started":"2022-08-12T09:00:29.658553Z","shell.execute_reply":"2022-08-12T09:00:33.077924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_len = len(test)\naverage_values = test.mean(numeric_only=True)\ntest = test.fillna(average_values)\nprint(f'training exemples : {len(test)},  {(1-len(test)/total_len)*100:.1f} % data not NaN')\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:00:33.080575Z","iopub.execute_input":"2022-08-12T09:00:33.080891Z","iopub.status.idle":"2022-08-12T09:00:33.217737Z","shell.execute_reply.started":"2022-08-12T09:00:33.080864Z","shell.execute_reply":"2022-08-12T09:00:33.216695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\nimport pickle\n# 保存模型\nwith open(\"./automl_v2.pkl\", \"wb\") as f:\n    pickle.dump(automl, f, pickle.HIGHEST_PROTOCOL)\n \n# 加载模型并预测\nwith open(\"./automl_v2.pkl\", \"rb\") as f:\n    automl = pickle.load(f)\n    \nX_test = test.drop([\"id\"],axis=1)\npred = automl.predict(X_test)\n\ntest[\"loss\"] = pred\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:00:33.219169Z","iopub.execute_input":"2022-08-12T09:00:33.219703Z","iopub.status.idle":"2022-08-12T09:00:34.655024Z","shell.execute_reply.started":"2022-08-12T09:00:33.219664Z","shell.execute_reply":"2022-08-12T09:00:34.653915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.index.name = 'id'","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:00:34.656693Z","iopub.execute_input":"2022-08-12T09:00:34.657342Z","iopub.status.idle":"2022-08-12T09:00:34.662216Z","shell.execute_reply.started":"2022-08-12T09:00:34.657277Z","shell.execute_reply":"2022-08-12T09:00:34.661358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(\n    {'id':test[\"id\"] ,\n     'loss': test[\"loss\"]},columns=['id', 'loss'])","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:01:07.293341Z","iopub.execute_input":"2022-08-12T09:01:07.293818Z","iopub.status.idle":"2022-08-12T09:01:07.314141Z","shell.execute_reply.started":"2022-08-12T09:01:07.293783Z","shell.execute_reply":"2022-08-12T09:01:07.312944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nsubmission.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T09:01:10.110871Z","iopub.execute_input":"2022-08-12T09:01:10.111601Z","iopub.status.idle":"2022-08-12T09:01:10.426119Z","shell.execute_reply.started":"2022-08-12T09:01:10.111565Z","shell.execute_reply":"2022-08-12T09:01:10.424990Z"},"trusted":true},"execution_count":null,"outputs":[]}]}