{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T07:36:36.676642Z","iopub.execute_input":"2022-07-14T07:36:36.677151Z","iopub.status.idle":"2022-07-14T07:36:36.713738Z","shell.execute_reply.started":"2022-07-14T07:36:36.677050Z","shell.execute_reply":"2022-07-14T07:36:36.712705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/restaurant-revenue-prediction/train.csv.zip\")\ndf_test = pd.read_csv(\"/kaggle/input/restaurant-revenue-prediction/test.csv.zip\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:56:23.284219Z","iopub.execute_input":"2022-07-14T07:56:23.284720Z","iopub.status.idle":"2022-07-14T07:56:23.780645Z","shell.execute_reply.started":"2022-07-14T07:56:23.284683Z","shell.execute_reply":"2022-07-14T07:56:23.779580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\nfrom sklearn.preprocessing import LabelEncoder\n\n#目的変数を抽出\nrevenue = df_train[\"revenue\"]\ndel df_train[\"revenue\"]\n\n#前処理がしやすい様に、trainとtestを結合\ndf_whole = pd.concat([df_train, df_test], axis=0)\n\n#Open Dateを年・月・日に分解\ndf_whole[\"Open Date\"] = pd.to_datetime(df_whole[\"Open Date\"])\ndf_whole[\"Year\"] = df_whole[\"Open Date\"].apply(lambda x:x.year)\ndf_whole[\"Month\"] = df_whole[\"Open Date\"].apply(lambda x:x.month)\ndf_whole[\"Day\"] = df_whole[\"Open Date\"].apply(lambda x:x.day)\n\n#Cityを数値に変換\nle = LabelEncoder()\ndf_whole[\"City\"] = le.fit_transform(df_whole[\"City\"])\n\n# City Groupを数値に変換 Other -> 0, Big Cities -> 1\ndf_whole[\"City Group\"] = df_whole[\"City Group\"].map({\"Other\":0, \"Big Cities\":1})\n\n#Typeを数値に変換 FC -> 0, IL -> 1, DT -> 2, MB -> 3\ndf_whole[\"Type\"] = df_whole[\"Type\"].map({\"FC\":0, \"IL\":1, \"DT\":2, \"MB\":3})\n\n#再びtrainとtestに分割\ndf_train = df_whole.iloc[:df_train.shape[0]]\ndf_test = df_whole.iloc[df_train.shape[0]:]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:57:00.370328Z","iopub.execute_input":"2022-07-14T07:57:00.370732Z","iopub.status.idle":"2022-07-14T07:57:03.083542Z","shell.execute_reply.started":"2022-07-14T07:57:00.370701Z","shell.execute_reply":"2022-07-14T07:57:03.082258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestRegressor\n\n#学習に使う特徴量を取得\ndf_train_columns = [col for col in df_train.columns if col not in [\"Id\", \"Open Date\"]]\n\n#RandomForestで学習させる\nrf = RandomForestRegressor(\n    n_estimators=200, \n    max_depth=5, \n    max_features=0.5, \n    random_state=449,\n    n_jobs=-1\n)\nrf.fit(df_train[df_train_columns], revenue)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:58:00.144470Z","iopub.execute_input":"2022-07-14T07:58:00.144917Z","iopub.status.idle":"2022-07-14T07:58:00.932336Z","shell.execute_reply.started":"2022-07-14T07:58:00.144884Z","shell.execute_reply":"2022-07-14T07:58:00.931258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = rf.predict(df_test[df_train_columns])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T07:58:51.513199Z","iopub.execute_input":"2022-07-14T07:58:51.513980Z","iopub.status.idle":"2022-07-14T07:58:52.012820Z","shell.execute_reply.started":"2022-07-14T07:58:51.513945Z","shell.execute_reply":"2022-07-14T07:58:52.011290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"Id\":df_test.Id, \"Prediction\":prediction})\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T08:12:20.019640Z","iopub.execute_input":"2022-07-14T08:12:20.020063Z","iopub.status.idle":"2022-07-14T08:12:20.391991Z","shell.execute_reply.started":"2022-07-14T08:12:20.020031Z","shell.execute_reply":"2022-07-14T08:12:20.390991Z"},"trusted":true},"execution_count":null,"outputs":[]}]}