{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n#importing the libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n# import seaborn as sns\nimport seaborn as sns; sns.set(style=\"ticks\", color_codes=True)\n\nfrom datetime import datetime\nfrom scipy import stats\nfrom scipy.stats import norm, skew\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport lightgbm as lgb\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:06.226935Z","iopub.execute_input":"2022-07-26T13:24:06.227354Z","iopub.status.idle":"2022-07-26T13:24:06.242418Z","shell.execute_reply.started":"2022-07-26T13:24:06.227322Z","shell.execute_reply":"2022-07-26T13:24:06.241323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:06.793436Z","iopub.execute_input":"2022-07-26T13:24:06.794573Z","iopub.status.idle":"2022-07-26T13:24:06.825449Z","shell.execute_reply.started":"2022-07-26T13:24:06.794526Z","shell.execute_reply":"2022-07-26T13:24:06.824284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ntrain = pd.read_csv('/kaggle/input/restaurant-revenue-prediction/train.csv.zip')\ntest = pd.read_csv('/kaggle/input/restaurant-revenue-prediction/test.csv.zip')\n\n# Idは不要なので、削除して別に変数化し、スコア提出時に使用\ntrain_Id = train.Id\ntest_Id = test.Id\n\n# Id列削除\ntrain.drop('Id', axis=1, inplace=True)\ntest.drop('Id', axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:07.409256Z","iopub.execute_input":"2022-07-26T13:24:07.410336Z","iopub.status.idle":"2022-07-26T13:24:07.916049Z","shell.execute_reply.started":"2022-07-26T13:24:07.410286Z","shell.execute_reply":"2022-07-26T13:24:07.914587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:07.918105Z","iopub.execute_input":"2022-07-26T13:24:07.918482Z","iopub.status.idle":"2022-07-26T13:24:07.960466Z","shell.execute_reply.started":"2022-07-26T13:24:07.918450Z","shell.execute_reply":"2022-07-26T13:24:07.959201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trci = train.groupby('City').count()\ntrci","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:34:05.217633Z","iopub.execute_input":"2022-07-26T13:34:05.218111Z","iopub.status.idle":"2022-07-26T13:34:05.274130Z","shell.execute_reply.started":"2022-07-26T13:34:05.218075Z","shell.execute_reply":"2022-07-26T13:34:05.273291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(trci.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:30:41.560034Z","iopub.execute_input":"2022-07-26T13:30:41.560450Z","iopub.status.idle":"2022-07-26T13:30:41.567304Z","shell.execute_reply.started":"2022-07-26T13:30:41.560417Z","shell.execute_reply":"2022-07-26T13:30:41.566229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pylab import rcParams\nrcParams['figure.figsize'] = 11, 4\ntrci[['Type']].groupby('City').mean().plot(kind='bar')\nplt.title('City Type of Training Data')\nplt.xlabel('City')\nplt.ylabel('City Count')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:07:29.738148Z","iopub.execute_input":"2022-07-26T14:07:29.738573Z","iopub.status.idle":"2022-07-26T14:07:30.173701Z","shell.execute_reply.started":"2022-07-26T14:07:29.738538Z","shell.execute_reply":"2022-07-26T14:07:30.172230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"teci = test.groupby('City').count()\nteci[['Type']].groupby('City').mean().plot(kind='bar')\nplt.title('City Type of Test Data')\nplt.xlabel('City')\nplt.ylabel('City Count')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T14:07:43.355799Z","iopub.execute_input":"2022-07-26T14:07:43.356209Z","iopub.status.idle":"2022-07-26T14:07:44.329643Z","shell.execute_reply.started":"2022-07-26T14:07:43.356174Z","shell.execute_reply":"2022-07-26T14:07:44.328259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 最大カラム数を100に拡張(デフォルトだと省略されてしまうので)\n# 常に全ての列（カラム）を表示\npd.options.display.max_columns = None\npd.options.display.max_rows = 80\n\n# 小数点2桁で表示(指数表記しないように)\npd.options.display.float_format = '{:.2f}'.format\n%matplotlib inline\n#ワーニングを抑止\nimport warnings\nwarnings.filterwarnings('ignore')\n%matplotlib inline\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:09.913217Z","iopub.execute_input":"2022-07-26T13:24:09.913631Z","iopub.status.idle":"2022-07-26T13:24:09.927287Z","shell.execute_reply.started":"2022-07-26T13:24:09.913596Z","shell.execute_reply":"2022-07-26T13:24:09.926151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Size of train data', train.shape)\nprint('Size of test data', test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:10.450340Z","iopub.execute_input":"2022-07-26T13:24:10.450712Z","iopub.status.idle":"2022-07-26T13:24:10.457146Z","shell.execute_reply.started":"2022-07-26T13:24:10.450682Z","shell.execute_reply":"2022-07-26T13:24:10.456008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:11.083349Z","iopub.execute_input":"2022-07-26T13:24:11.083728Z","iopub.status.idle":"2022-07-26T13:24:11.091384Z","shell.execute_reply.started":"2022-07-26T13:24:11.083698Z","shell.execute_reply":"2022-07-26T13:24:11.090155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:11.632361Z","iopub.execute_input":"2022-07-26T13:24:11.632760Z","iopub.status.idle":"2022-07-26T13:24:11.651050Z","shell.execute_reply.started":"2022-07-26T13:24:11.632721Z","shell.execute_reply":"2022-07-26T13:24:11.649519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:12.257976Z","iopub.execute_input":"2022-07-26T13:24:12.258375Z","iopub.status.idle":"2022-07-26T13:24:12.363385Z","shell.execute_reply.started":"2022-07-26T13:24:12.258345Z","shell.execute_reply":"2022-07-26T13:24:12.362243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include='O')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:13.067807Z","iopub.execute_input":"2022-07-26T13:24:13.068210Z","iopub.status.idle":"2022-07-26T13:24:13.090191Z","shell.execute_reply.started":"2022-07-26T13:24:13.068178Z","shell.execute_reply":"2022-07-26T13:24:13.089049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"revenue\"].describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:14.986295Z","iopub.execute_input":"2022-07-26T13:24:14.986678Z","iopub.status.idle":"2022-07-26T13:24:14.997665Z","shell.execute_reply.started":"2022-07-26T13:24:14.986647Z","shell.execute_reply":"2022-07-26T13:24:14.996669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#目的変数であるrevenueのヒストグラムとQ-Qプロットを表示する\n# 分布確認\nfig = plt.figure(figsize=(10, 4))\nplt.subplots_adjust(wspace=0.4)\n\n# ヒストグラム\nax = fig.add_subplot(1, 2, 1)\nsns.distplot(train['revenue'], ax=ax)\n\n# QQプロット\nax2 = fig.add_subplot(1, 2, 2)\nstats.probplot(train['revenue'], plot=ax2)\n\nplt.show()\n\n# 変換後の要約統計量表示\nprint(train['revenue'].describe())\nprint(\"------------------------------\")\nprint(\"歪度: %f\" % train['revenue'].skew())\nprint(\"尖度: %f\" % train['revenue'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:46.925053Z","iopub.execute_input":"2022-07-26T13:24:46.925445Z","iopub.status.idle":"2022-07-26T13:24:48.304499Z","shell.execute_reply.started":"2022-07-26T13:24:46.925414Z","shell.execute_reply":"2022-07-26T13:24:48.303303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データをコピーし、新たなdataframeで検証\ndf = train.copy()\n\n#目的変数の対数log(x+1)をとる\ndf['revenue'] = np.log1p(df['revenue'])\n\n# 標準化(平均0, 分散1)\nscaler=StandardScaler()\ndf['revenue']=scaler.fit_transform(df[['revenue']])\n\n# 分布確認\nfig = plt.figure(figsize=(10, 4))\nplt.subplots_adjust(wspace=0.4)\n\n# ヒストグラム\nax = fig.add_subplot(1, 2, 1)\nsns.distplot(df['revenue'], ax=ax)\n\n# QQプロット\nax2 = fig.add_subplot(1, 2, 2)\nstats.probplot(df['revenue'], plot=ax2)\n\nplt.show()\n\n# 変換後の要約統計量表示\nprint(df['revenue'].describe())\nprint(\"------------------------------\")\nprint(\"歪度: %f\" % df['revenue'].skew())\nprint(\"尖度: %f\" % df['revenue'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:49.727453Z","iopub.execute_input":"2022-07-26T13:24:49.727884Z","iopub.status.idle":"2022-07-26T13:24:50.049525Z","shell.execute_reply.started":"2022-07-26T13:24:49.727849Z","shell.execute_reply":"2022-07-26T13:24:50.047864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データをコピーし、新たなdataframeで検証\ndf = train.copy()\n\n# 標準化(平均0, 分散1)\nscaler=StandardScaler()\ndf['revenue']=scaler.fit_transform(df[['revenue']])\n\n\n# 分布確認\nfig = plt.figure(figsize=(10, 4))\nplt.subplots_adjust(wspace=0.4)\n\n# ヒストグラム\nax = fig.add_subplot(1, 2, 1)\nsns.distplot(df['revenue'], ax=ax)\n\n# QQプロット\nax2 = fig.add_subplot(1, 2, 2)\nstats.probplot(df['revenue'], plot=ax2)\n\nplt.show()\n\n# 変換後の要約統計量表示\nprint(df['revenue'].describe())\nprint(\"------------------------------\")\nprint(\"歪度: %f\" % df['revenue'].skew())\nprint(\"尖度: %f\" % df['revenue'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:50.864443Z","iopub.execute_input":"2022-07-26T13:24:50.864858Z","iopub.status.idle":"2022-07-26T13:24:51.240577Z","shell.execute_reply.started":"2022-07-26T13:24:50.864823Z","shell.execute_reply":"2022-07-26T13:24:51.239352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データをコピーし、新たなdataframeで検証\ndf = train.copy()\n\n# Min-Max変換(正規化(最大1, 最小0))\nscaler=MinMaxScaler()\ndf['revenue']=scaler.fit_transform(df[['revenue']])\n\n# 分布確認\nfig = plt.figure(figsize=(10, 4))\nplt.subplots_adjust(wspace=0.4)\n\n# ヒストグラム\nax = fig.add_subplot(1, 2, 1)\nsns.distplot(df['revenue'], ax=ax)\n\n# QQプロット\nax2 = fig.add_subplot(1, 2, 2)\nstats.probplot(df['revenue'], plot=ax2)\n\nplt.show()\n\n# 変換後の要約統計量表示\nprint(df['revenue'].describe())\nprint(\"------------------------------\")\nprint(\"歪度: %f\" % df['revenue'].skew())\nprint(\"尖度: %f\" % df['revenue'].kurt())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:52.036688Z","iopub.execute_input":"2022-07-26T13:24:52.037737Z","iopub.status.idle":"2022-07-26T13:24:52.352161Z","shell.execute_reply.started":"2022-07-26T13:24:52.037687Z","shell.execute_reply":"2022-07-26T13:24:52.350231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データ\n# Open Dateを日付型に変換\ntrain['pd_date'] = pd.to_datetime(train['Open Date'], format='%m/%d/%Y')\n# 年のみを抽出\ntrain['Open_Year'] = train['pd_date'].dt.strftime('%Y')\n# 月のみを抽出\ntrain['Open_Month'] = train['pd_date'].dt.strftime('%m')\n# 経過年数\ntrain[\"Open Date\"] = pd.to_datetime(train[\"Open Date\"])\ntrain[\"Day\"] = train[\"Open Date\"].apply(lambda x:x.day)\ntrain[\"kijun\"] = \"2022-07-10\"\ntrain[\"kijun\"] = pd.to_datetime(train[\"kijun\"])\ntrain[\"BusinessPeriod\"] = (train[\"kijun\"] - train[\"Open Date\"]).apply(lambda x: x.days)\n\ntrain = train.drop('kijun', axis=1)\n\ntrain = train.drop('pd_date',axis=1)\ntrain = train.drop('Open Date',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:53.179580Z","iopub.execute_input":"2022-07-26T13:24:53.180840Z","iopub.status.idle":"2022-07-26T13:24:53.209802Z","shell.execute_reply.started":"2022-07-26T13:24:53.180794Z","shell.execute_reply":"2022-07-26T13:24:53.208705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータ\n# Open Dateを日付型に変換\ntest['pd_date'] = pd.to_datetime(test['Open Date'], format='%m/%d/%Y')\n# 年のみを抽出\ntest['Open_Year'] = test['pd_date'].dt.strftime('%Y')\n# 月のみを抽出\ntest['Open_Month'] = test['pd_date'].dt.strftime('%m')\n# 経過年数\ntest[\"Open Date\"] = pd.to_datetime(test[\"Open Date\"])\ntest[\"Day\"] = test[\"Open Date\"].apply(lambda x:x.day)\ntest[\"kijun\"] = \"2022-07-10\"\ntest[\"kijun\"] = pd.to_datetime(test[\"kijun\"])\ntest[\"BusinessPeriod\"] = (test[\"kijun\"] - test[\"Open Date\"]).apply(lambda x: x.days)\n\ntest = test.drop('kijun', axis=1)\ntest = test.drop('pd_date',axis=1)\ntest = test.drop('Open Date',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:24:58.574812Z","iopub.execute_input":"2022-07-26T13:24:58.575641Z","iopub.status.idle":"2022-07-26T13:25:01.739296Z","shell.execute_reply.started":"2022-07-26T13:24:58.575592Z","shell.execute_reply":"2022-07-26T13:25:01.738318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.dtypes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:01.740843Z","iopub.execute_input":"2022-07-26T13:25:01.741380Z","iopub.status.idle":"2022-07-26T13:25:01.750259Z","shell.execute_reply.started":"2022-07-26T13:25:01.741347Z","shell.execute_reply":"2022-07-26T13:25:01.749105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#カテゴリ変数と数値変数に分ける\ncats = list(train.select_dtypes(include=['object']).columns)\nnums = list(train.select_dtypes(exclude=['object']).columns)\nprint(f'categorical variables:  {cats}')\nprint(f'numerical variables:  {nums}')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:03.394940Z","iopub.execute_input":"2022-07-26T13:25:03.395349Z","iopub.status.idle":"2022-07-26T13:25:03.406201Z","shell.execute_reply.started":"2022-07-26T13:25:03.395316Z","shell.execute_reply":"2022-07-26T13:25:03.404805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.nunique(axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:04.253602Z","iopub.execute_input":"2022-07-26T13:25:04.254067Z","iopub.status.idle":"2022-07-26T13:25:04.272702Z","shell.execute_reply.started":"2022-07-26T13:25:04.254034Z","shell.execute_reply":"2022-07-26T13:25:04.271139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'categorical variables:  {cats}')\nprint(f'numerical variables:  {nums}')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:05.016552Z","iopub.execute_input":"2022-07-26T13:25:05.017264Z","iopub.status.idle":"2022-07-26T13:25:05.023251Z","shell.execute_reply.started":"2022-07-26T13:25:05.017226Z","shell.execute_reply":"2022-07-26T13:25:05.022057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 名義変数\nnominal_list =cats\n               \n# 順序変数\n# ordinal_list = []\n\n# 数値変数\nnum_list = nums\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:08.887387Z","iopub.execute_input":"2022-07-26T13:25:08.887820Z","iopub.status.idle":"2022-07-26T13:25:08.892758Z","shell.execute_reply.started":"2022-07-26T13:25:08.887766Z","shell.execute_reply":"2022-07-26T13:25:08.891691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = int(len(nominal_list)/2+1)\n\nfig = plt.figure(figsize=(30, 20))\nplt.subplots_adjust(hspace=0.6, wspace=0.4)\n\nfor i in range(len(nominal_list)):\n    ax = fig.add_subplot(columns, 2, i+1)\n    sns.countplot(x=nominal_list[i], data=train, ax=ax)\n    plt.xticks(rotation=45)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:10.095034Z","iopub.execute_input":"2022-07-26T13:25:10.095437Z","iopub.status.idle":"2022-07-26T13:25:11.177832Z","shell.execute_reply.started":"2022-07-26T13:25:10.095403Z","shell.execute_reply":"2022-07-26T13:25:11.176693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = int(len(num_list)/3+1)\n\nfig = plt.figure(figsize=(30, 40))\nplt.subplots_adjust(hspace=0.6, wspace=0.4)\n\nfor i in range(len(num_list)):\n    ax = fig.add_subplot(columns, 3, i+1)\n\n    train[num_list[i]].hist(ax=ax)\n    ax2 = train[num_list[i]].plot.kde(ax=ax, secondary_y=True,title=num_list[i])\n    ax2.set_ylim(0)\n    \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:11.180542Z","iopub.execute_input":"2022-07-26T13:25:11.181024Z","iopub.status.idle":"2022-07-26T13:25:20.085909Z","shell.execute_reply.started":"2022-07-26T13:25:11.180978Z","shell.execute_reply":"2022-07-26T13:25:20.084026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = int(len(nominal_list)/2+1)\n\nfig = plt.figure(figsize=(20, 10))\nplt.subplots_adjust(hspace=0.6, wspace=0.4)\n\nfor i in range(len(nominal_list)):\n    ax = fig.add_subplot(columns, 2, i+1)\n\n    # 回帰の場合    \n    sns.boxplot(x=nominal_list[i], y=train.revenue, data=train, ax=ax)\n    plt.xticks(rotation=45)\n    # 分類の場合\n#     sns.barplot(x = nominal_list[i], y = train.revenue, data=train, ax=ax)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:20.088502Z","iopub.execute_input":"2022-07-26T13:25:20.090001Z","iopub.status.idle":"2022-07-26T13:25:21.775622Z","shell.execute_reply.started":"2022-07-26T13:25:20.089950Z","shell.execute_reply":"2022-07-26T13:25:21.774236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.drop('Open_Month',axis=1)\ntest= test.drop('Open_Month',axis=1)\nnominal_list.remove('Open_Month')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:21.777597Z","iopub.execute_input":"2022-07-26T13:25:21.778654Z","iopub.status.idle":"2022-07-26T13:25:21.800503Z","shell.execute_reply.started":"2022-07-26T13:25:21.778617Z","shell.execute_reply":"2022-07-26T13:25:21.799153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = int(len(num_list)/4+1)\n\nfig = plt.figure(figsize=(30, 35))\nplt.subplots_adjust(hspace=0.6, wspace=0.4)\n\nfor i in range(len(num_list)):\n    ax = fig.add_subplot(columns, 4, i+1)\n\n    # 回帰の場合    \n    sns.regplot(x=num_list[i],y='revenue',data=train, ax=ax)\n    plt.xticks(rotation=45)\n    # 分類の場合\n#     sns.barplot(x = nominal_list[i], y = train.revenue, data=train, ax=ax)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:21.803657Z","iopub.execute_input":"2022-07-26T13:25:21.804199Z","iopub.status.idle":"2022-07-26T13:25:31.836809Z","shell.execute_reply.started":"2022-07-26T13:25:21.804154Z","shell.execute_reply":"2022-07-26T13:25:31.835873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['City','revenue']].groupby('City').mean().plot(kind='bar')\nplt.title('Mean Revenue Generated vs City')\nplt.xlabel('City')\nplt.ylabel('Mean Revenue Generated')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:25:31.838068Z","iopub.execute_input":"2022-07-26T13:25:31.838952Z","iopub.status.idle":"2022-07-26T13:25:32.236308Z","shell.execute_reply.started":"2022-07-26T13:25:31.838915Z","shell.execute_reply":"2022-07-26T13:25:32.235041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cityごとのrevenue平均値を1000000単位とする\nmean_revenue_per_city = train[['City', 'revenue']].groupby('City', as_index=False).mean()\nmean_revenue_per_city.head()\nmean_revenue_per_city['revenue'] = mean_revenue_per_city['revenue'].apply(lambda x: int(x/1e6)) \n\nmean_revenue_per_city\n\nmean_dict = dict(zip(mean_revenue_per_city.City, mean_revenue_per_city.revenue))\nmean_dict","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:41.959704Z","iopub.execute_input":"2022-07-26T13:19:41.960577Z","iopub.status.idle":"2022-07-26T13:19:41.978086Z","shell.execute_reply.started":"2022-07-26T13:19:41.960528Z","shell.execute_reply":"2022-07-26T13:19:41.977255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train['City'].sort_values().unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:41.979559Z","iopub.execute_input":"2022-07-26T13:19:41.980567Z","iopub.status.idle":"2022-07-26T13:19:41.988627Z","shell.execute_reply.started":"2022-07-26T13:19:41.980529Z","shell.execute_reply":"2022-07-26T13:19:41.986958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['City'].sort_values().unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:41.990916Z","iopub.execute_input":"2022-07-26T13:19:41.992224Z","iopub.status.idle":"2022-07-26T13:19:42.146072Z","shell.execute_reply.started":"2022-07-26T13:19:41.992161Z","shell.execute_reply":"2022-07-26T13:19:42.144818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cityについて、学習データとテストデータにて重複削除し、リスト化\ncity_train_list = list(train['City'].unique())\ncity_test_list = list(test['City'].unique())","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.147481Z","iopub.execute_input":"2022-07-26T13:19:42.148502Z","iopub.status.idle":"2022-07-26T13:19:42.163845Z","shell.execute_reply.started":"2022-07-26T13:19:42.148462Z","shell.execute_reply":"2022-07-26T13:19:42.162531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"l1_l2_and = set(city_train_list) & set(city_test_list)\nprint(l1_l2_and)\nprint(len(l1_l2_and))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.165658Z","iopub.execute_input":"2022-07-26T13:19:42.166268Z","iopub.status.idle":"2022-07-26T13:19:42.172581Z","shell.execute_reply.started":"2022-07-26T13:19:42.166232Z","shell.execute_reply":"2022-07-26T13:19:42.171544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# どちらかにしかないCityを抽出\nl1_l2_sym_diff = set(city_test_list) ^ set(city_train_list)\nprint(l1_l2_sym_diff)\nprint(len(l1_l2_sym_diff))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.174323Z","iopub.execute_input":"2022-07-26T13:19:42.175067Z","iopub.status.idle":"2022-07-26T13:19:42.188622Z","shell.execute_reply.started":"2022-07-26T13:19:42.175029Z","shell.execute_reply":"2022-07-26T13:19:42.187426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータのみ存在するCityの件数\nlen(set(city_test_list).difference(city_train_list))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.190253Z","iopub.execute_input":"2022-07-26T13:19:42.190627Z","iopub.status.idle":"2022-07-26T13:19:42.208474Z","shell.execute_reply.started":"2022-07-26T13:19:42.190595Z","shell.execute_reply":"2022-07-26T13:19:42.207041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データのみ存在するCityの件数\nlen(set(city_train_list).difference(city_test_list))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.209765Z","iopub.execute_input":"2022-07-26T13:19:42.210847Z","iopub.status.idle":"2022-07-26T13:19:42.221766Z","shell.execute_reply.started":"2022-07-26T13:19:42.210805Z","shell.execute_reply":"2022-07-26T13:19:42.220628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# P変数の1つのクラスは地理的属性であると指定されているため\n# 各都市のP変数の平均をプロットすると、どのP変数が都市と関連性が高いかが分かる\ndistinct_cities = train.loc[:, \"City\"].unique()\n\n# P変数のcityごとの平均値を取得\nmeans = []\nfor i in range(len(num_list)):\n    temp = []\n    for city in distinct_cities:\n        temp.append(train.loc[train.City == city, num_list[i]].mean())  \n    means.append(temp)\n    \ncity_pvars = pd.DataFrame(columns=[\"city_var\", \"means\"])\nfor i in range(37):\n    for j in range(len(distinct_cities)):\n        city_pvars.loc[i+37*j] = [\"P\"+str(i+1), means[i][j]]\n\nprint(city_pvars)            \n# 箱ひげ図を表示\nplt.rcParams['figure.figsize'] = (18.0, 6.0)\nsns.boxplot(x=\"city_var\", y=\"means\", data=city_pvars)\n\n# From this we observe that P1, P2, P11, P19, P20, P23, and P30 are approximately a good\n# proxy for geographical location.\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:42.223227Z","iopub.execute_input":"2022-07-26T13:19:42.224407Z","iopub.status.idle":"2022-07-26T13:19:46.262164Z","shell.execute_reply.started":"2022-07-26T13:19:42.224365Z","shell.execute_reply":"2022-07-26T13:19:46.260760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import cluster\n\ndef adjust_cities(full_full_data, train, k):\n    \n    # As found by box plot of each city's mean over each p-var\n    relevant_pvars =  [\"P1\", \"P2\", \"P11\", \"P19\", \"P20\", \"P23\",\"P30\"]\n    train = train.loc[:, relevant_pvars]\n    \n    # Optimal k is 20 as found by DB-Index plot    \n    kmeans = cluster.KMeans(n_clusters=k)\n    kmeans.fit(train)\n    \n    # Get the cluster centers and classify city of each full_data instance to one of the centers\n    full_data['City_Cluster'] = kmeans.predict(full_data.loc[:, relevant_pvars])\n    \n    return full_data","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:46.265715Z","iopub.execute_input":"2022-07-26T13:19:46.266753Z","iopub.status.idle":"2022-07-26T13:19:46.529966Z","shell.execute_reply.started":"2022-07-26T13:19:46.266712Z","shell.execute_reply":"2022-07-26T13:19:46.528572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_train = train.shape[0]\nnum_test = test.shape[0]\nprint(num_train, num_test)\n\nfull_data = pd.concat([train, test], ignore_index=True)  ","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:46.531632Z","iopub.execute_input":"2022-07-26T13:19:46.532039Z","iopub.status.idle":"2022-07-26T13:19:46.568639Z","shell.execute_reply.started":"2022-07-26T13:19:46.532005Z","shell.execute_reply":"2022-07-26T13:19:46.567555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 学習データを使用しクラスタリングを行い、その学習結果を全データに適用させる\nfull_data = adjust_cities(full_data, train, 20)\nfull_data\n\n# City項目は不要なので削除\nfull_data = full_data.drop(['City'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:46.569977Z","iopub.execute_input":"2022-07-26T13:19:46.570577Z","iopub.status.idle":"2022-07-26T13:19:46.883671Z","shell.execute_reply.started":"2022-07-26T13:19:46.570542Z","shell.execute_reply":"2022-07-26T13:19:46.880534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split into train and test datasets\ntrain = full_data[:num_train]\ntest = full_data[num_train:]\n# check the shapes \nprint(\"Train :\",train.shape)\nprint(\"Test:\",test.shape)\ntest","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:46.886123Z","iopub.execute_input":"2022-07-26T13:19:46.887065Z","iopub.status.idle":"2022-07-26T13:19:46.941791Z","shell.execute_reply.started":"2022-07-26T13:19:46.887009Z","shell.execute_reply":"2022-07-26T13:19:46.939948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[['City_Cluster','revenue']].groupby('City_Cluster').mean().plot(kind='bar')\nplt.title('Mean Revenue Generated vs City Cluster')\nplt.xlabel('City Cluster')\nplt.ylabel('Mean Revenue Generated')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:46.944373Z","iopub.execute_input":"2022-07-26T13:19:46.945536Z","iopub.status.idle":"2022-07-26T13:19:47.277691Z","shell.execute_reply.started":"2022-07-26T13:19:46.945490Z","shell.execute_reply":"2022-07-26T13:19:47.276819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_revenue_per_city = train[['City_Cluster', 'revenue']].groupby('City_Cluster', as_index=False).mean()\nmean_revenue_per_city.head()\nmean_revenue_per_city['revenue'] = mean_revenue_per_city['revenue'].apply(lambda x: int(x/1e6)) \n\nmean_revenue_per_city\n\nmean_dict = dict(zip(mean_revenue_per_city.City_Cluster, mean_revenue_per_city.revenue))\nmean_dict","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.279157Z","iopub.execute_input":"2022-07-26T13:19:47.279728Z","iopub.status.idle":"2022-07-26T13:19:47.294822Z","shell.execute_reply.started":"2022-07-26T13:19:47.279695Z","shell.execute_reply":"2022-07-26T13:19:47.293675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"city_rev = []\n\nfor i in full_data['City_Cluster']:\n    for key, value in mean_dict.items():\n        if i == key:\n            city_rev.append(value)\n            \ndf_city_rev = pd.DataFrame({'city_rev':city_rev})\nfull_data = pd.concat([full_data,df_city_rev],axis=1)\nfull_data.head\n\n# 値の追加\nnominal_list.extend(['City_Cluster'])\n# 値の削除\nnominal_list.remove('City')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.296695Z","iopub.execute_input":"2022-07-26T13:19:47.297235Z","iopub.status.idle":"2022-07-26T13:19:47.725701Z","shell.execute_reply.started":"2022-07-26T13:19:47.297191Z","shell.execute_reply":"2022-07-26T13:19:47.724717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\nle_count = 0\n\n# Iterate through the columns\n# for col in application_full_data:\nfor i in range(len(nominal_list)):    \n    \n#     if application_full_data[col].dtype == 'object':\n        # If 2 or fewer unique categories\n        if len(list(full_data[nominal_list[i]].unique())) <= 2:\n            # full_data on the full_dataing data\n            le.fit(full_data[nominal_list[i]])\n            # Transform both full_dataing and testing data\n            full_data[nominal_list[i]] = le.transform(full_data[nominal_list[i]])\n            \n            # Keep track of how many columns were label encoded\n            le_count += 1\n            \nprint('%d columns were label encoded.' % le_count)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.727084Z","iopub.execute_input":"2022-07-26T13:19:47.728231Z","iopub.status.idle":"2022-07-26T13:19:47.800969Z","shell.execute_reply.started":"2022-07-26T13:19:47.728188Z","shell.execute_reply":"2022-07-26T13:19:47.799704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encoding of categorical variables\nfull_data = pd.get_dummies(full_data)\nprint('full_dataing Features shape: ', full_data.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.802304Z","iopub.execute_input":"2022-07-26T13:19:47.802682Z","iopub.status.idle":"2022-07-26T13:19:47.888909Z","shell.execute_reply.started":"2022-07-26T13:19:47.802635Z","shell.execute_reply":"2022-07-26T13:19:47.887562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tukey_outliers(x):\n    q1 = np.percentile(x,25)\n    q3 = np.percentile(x,75)\n    \n    iqr = q3-q1\n    \n    min_range = q1 - iqr*1.5\n    max_range = q3 + iqr*1.5\n    \n    outliers = x[(x<min_range) | (x>max_range)]\n    return outliers","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.890855Z","iopub.execute_input":"2022-07-26T13:19:47.891534Z","iopub.status.idle":"2022-07-26T13:19:47.899086Z","shell.execute_reply.started":"2022-07-26T13:19:47.891485Z","shell.execute_reply":"2022-07-26T13:19:47.898107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = int(len(num_list)/4+1)\n\n# boxplot\nfig = plt.figure(figsize=(15,20))\nplt.subplots_adjust(hspace=0.2, wspace=0.8)\nfor i in range(len(num_list)):\n    ax = fig.add_subplot(columns, 4, i+1)\n    sns.boxplot(y=full_data[num_list[i]], data=full_data, ax=ax)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:47.900473Z","iopub.execute_input":"2022-07-26T13:19:47.901082Z","iopub.status.idle":"2022-07-26T13:19:53.061812Z","shell.execute_reply.started":"2022-07-26T13:19:47.901049Z","shell.execute_reply":"2022-07-26T13:19:53.060425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skewed_data = train[num_list].apply(lambda x: skew(x)).sort_values(ascending=False)\nskewed_data[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:53.063634Z","iopub.execute_input":"2022-07-26T13:19:53.064142Z","iopub.status.idle":"2022-07-26T13:19:53.086589Z","shell.execute_reply.started":"2022-07-26T13:19:53.064094Z","shell.execute_reply":"2022-07-26T13:19:53.085716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split into train and test datasets\ntrain = full_data[:num_train]\ntest = full_data[num_train:]\n# check the shapes \nprint(\"Train :\",train.shape)\nprint(\"Test:\",test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:53.087890Z","iopub.execute_input":"2022-07-26T13:19:53.088394Z","iopub.status.idle":"2022-07-26T13:19:53.094535Z","shell.execute_reply.started":"2022-07-26T13:19:53.088361Z","shell.execute_reply":"2022-07-26T13:19:53.093591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(font_scale=1.1)\ncorrelation_train = train.corr()\nmask = np.triu(correlation_train.corr())\nfig = plt.figure(figsize=(50,50))\nsns.heatmap(correlation_train,\n            annot=True,\n            fmt='.1f',\n            cmap='coolwarm',\n            square=True,\n#             mask=mask,\n            linewidths=1)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:19:53.096098Z","iopub.execute_input":"2022-07-26T13:19:53.096637Z","iopub.status.idle":"2022-07-26T13:20:09.431241Z","shell.execute_reply.started":"2022-07-26T13:19:53.096606Z","shell.execute_reply":"2022-07-26T13:20:09.429941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find correlations with the target and sort\ncorrelations = train.corr()['revenue'].sort_values()\n\n# Display correlations\nprint('Most Positive Correlations:\\n', correlations.tail(15))\nprint('\\nMost Negative Correlations:\\n', correlations.head(15))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.433225Z","iopub.execute_input":"2022-07-26T13:20:09.433861Z","iopub.status.idle":"2022-07-26T13:20:09.449741Z","shell.execute_reply.started":"2022-07-26T13:20:09.433810Z","shell.execute_reply":"2022-07-26T13:20:09.448835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 相関が高い10項目のみ抽出\ncorrelations = train.corr()\n# 絶対値で取得\ncorrelations = abs(correlations)\n\ncols = correlations.nlargest(10,'revenue')['revenue'].index\ncols","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.451348Z","iopub.execute_input":"2022-07-26T13:20:09.452483Z","iopub.status.idle":"2022-07-26T13:20:09.471186Z","shell.execute_reply.started":"2022-07-26T13:20:09.452433Z","shell.execute_reply":"2022-07-26T13:20:09.469395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 相関が高い10項目のみ抽出\ntrain = train[cols]\n\n#学習データを目的変数とそれ以外に分ける\ntrain_X = train.drop(\"revenue\",axis=1)\ntrain_y = train[\"revenue\"]\n\n#revenueを対数変換する \ntrain_y = np.log1p(train_y)\n\n#テストデータを学習データのカラムのみにする \ntmp_cols = train_X.columns\ntest_X = test[tmp_cols]\n\n#それぞれのデータのサイズを確認\nprint(\"train_X: \"+str(train_X.shape))\nprint(\"train_y: \"+str(train_y.shape))\nprint(\"test_X: \"+str(test_X.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.473623Z","iopub.execute_input":"2022-07-26T13:20:09.474177Z","iopub.status.idle":"2022-07-26T13:20:09.522639Z","shell.execute_reply.started":"2022-07-26T13:20:09.474128Z","shell.execute_reply":"2022-07-26T13:20:09.521458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#訓練データとモデル評価用データに分けるライブラリ\nfrom sklearn.model_selection import train_test_split\n\n#フォールドアウト法により、学習データとテストデータに分割 \n(train_x, valid_x, train_y, valid_y) = train_test_split(train_X, train_y , test_size = 0.3 , random_state = 0)\n\nprint(\"X_train: \"+str(train_x.shape))\nprint(\"X_test: \"+str(valid_x.shape))\nprint(\"y_train: \"+str(train_y.shape))\nprint(\"y_test: \"+str(valid_y.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.524585Z","iopub.execute_input":"2022-07-26T13:20:09.525377Z","iopub.status.idle":"2022-07-26T13:20:09.536274Z","shell.execute_reply.started":"2022-07-26T13:20:09.525331Z","shell.execute_reply":"2022-07-26T13:20:09.535013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVR\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.ensemble import GradientBoostingRegressor\nfrom sklearn.ensemble import AdaBoostRegressor\nfrom sklearn.tree import DecisionTreeRegressor\nfrom xgboost import XGBRegressor\nimport xgboost as xgb\n\n# lightGBMによる予測\nlgb_train = lgb.Dataset(train_x, train_y)\nlgb_eval = lgb.Dataset(valid_x, valid_y, reference=lgb_train)\n\n# LightGBM parameters\nparams = {\n        'task' : 'train',\n        'boosting_type' : 'gbdt',\n        'objective' : 'regression',\n        'metric' : {'l2'},\n        'num_leaves' : 101,\n        'learning_rate' : 0.6,\n        'feature_fraction' : 0.9,\n        'bagging_fraction' : 0.8,\n        'bagging_freq': 5,\n        'verbose' : 0,\n        'n_jobs': 2\n}\n\ngbm = lgb.train(params,\n            lgb_train,\n            num_boost_round=100,\n            valid_sets=lgb_eval,\n            early_stopping_rounds=10)\n\nprediction_lgb = np.exp(gbm.predict(test_X))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.537935Z","iopub.execute_input":"2022-07-26T13:20:09.538883Z","iopub.status.idle":"2022-07-26T13:20:09.798655Z","shell.execute_reply.started":"2022-07-26T13:20:09.538834Z","shell.execute_reply":"2022-07-26T13:20:09.797396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 予測した値を提出用CSVファイル(submissionファイル)に書き出し\nsubmission = pd.DataFrame({\"Id\":test_Id, \"Prediction\":prediction_lgb})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T13:20:09.800141Z","iopub.execute_input":"2022-07-26T13:20:09.801187Z","iopub.status.idle":"2022-07-26T13:20:10.148532Z","shell.execute_reply.started":"2022-07-26T13:20:09.801150Z","shell.execute_reply":"2022-07-26T13:20:10.147043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}