{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-16T05:36:57.129587Z","iopub.execute_input":"2022-06-16T05:36:57.130408Z","iopub.status.idle":"2022-06-16T05:36:57.159375Z","shell.execute_reply.started":"2022-06-16T05:36:57.130232Z","shell.execute_reply":"2022-06-16T05:36:57.158405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing, impute, model_selection\nfrom datetime import date","metadata":{"execution":{"iopub.status.busy":"2022-06-16T05:36:57.160973Z","iopub.execute_input":"2022-06-16T05:36:57.161333Z","iopub.status.idle":"2022-06-16T05:36:58.528837Z","shell.execute_reply.started":"2022-06-16T05:36:57.161302Z","shell.execute_reply":"2022-06-16T05:36:58.528131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\",index_col=\"PassengerId\")\ntest_df  = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\", index_col=\"PassengerId\")\ndisplay(train_df.head())\ntrain_df[\"PassengerId\"] = train_df.index\ntest_df[\"PassengerId\"]  = test_df.index","metadata":{"execution":{"iopub.status.busy":"2022-06-16T05:37:13.074134Z","iopub.execute_input":"2022-06-16T05:37:13.074488Z","iopub.status.idle":"2022-06-16T05:37:13.204366Z","shell.execute_reply.started":"2022-06-16T05:37:13.074457Z","shell.execute_reply":"2022-06-16T05:37:13.203341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.495177Z","iopub.execute_input":"2022-06-16T02:36:26.496168Z","iopub.status.idle":"2022-06-16T02:36:26.535962Z","shell.execute_reply.started":"2022-06-16T02:36:26.496118Z","shell.execute_reply":"2022-06-16T02:36:26.534862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 特徴量エンジニアリング\n- `PassengerId`は`GroupId`と`GroupSize`に分けられそう","metadata":{}},{"cell_type":"code","source":"def from_passengerId(df):\n    split_id = df[\"PassengerId\"].str.split(\"_\",expand=True)\n    df[\"GroupId\"]   = split_id[0]\n    df[\"GroupSize\"] = df.groupby(\"GroupId\")[\"GroupId\"].transform(\"count\")\n\n    df[\"Alone\"] = (df[\"GroupSize\"] == 1)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.539616Z","iopub.execute_input":"2022-06-16T02:36:26.540215Z","iopub.status.idle":"2022-06-16T02:36:26.550878Z","shell.execute_reply.started":"2022-06-16T02:36:26.540152Z","shell.execute_reply":"2022-06-16T02:36:26.549342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = from_passengerId(train_df)\ntest_df  = from_passengerId(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.553373Z","iopub.execute_input":"2022-06-16T02:36:26.553895Z","iopub.status.idle":"2022-06-16T02:36:26.602428Z","shell.execute_reply.started":"2022-06-16T02:36:26.553851Z","shell.execute_reply":"2022-06-16T02:36:26.600934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 欠損値の確認\nお金の支出に関するカラムに対して欠損値があるかどうかを示す`_missing`カラムを追加する","metadata":{}},{"cell_type":"code","source":"def missing_value_features(df,columns,expenditure_columns):\n    for column in columns:\n        df[f\"{column}_missing\"] = df[column].isna()\n    \n    df[\"TotalExpense_missing\"] = df[expenditure_columns].sum(axis=1,skipna=False).isna()\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.603982Z","iopub.execute_input":"2022-06-16T02:36:26.604538Z","iopub.status.idle":"2022-06-16T02:36:26.609322Z","shell.execute_reply.started":"2022-06-16T02:36:26.604457Z","shell.execute_reply":"2022-06-16T02:36:26.608584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expenditure_columns = [\"RoomService\",\"FoodCourt\",\"ShoppingMall\",\"Spa\",\"VRDeck\"]\ncolumns = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"Cabin\", \"VIP\"]\ntrain_df = missing_value_features(train_df, columns, expenditure_columns)\ntest_df  = missing_value_features(test_df,  columns, expenditure_columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.610581Z","iopub.execute_input":"2022-06-16T02:36:26.611364Z","iopub.status.idle":"2022-06-16T02:36:26.641472Z","shell.execute_reply.started":"2022-06-16T02:36:26.611317Z","shell.execute_reply":"2022-06-16T02:36:26.640305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TotalExpense\nすべての支出カラムからその合計を抽出","metadata":{}},{"cell_type":"code","source":"def from_expenditure_feantures(df,expenditure_columns):\n    df[\"TotalExpense\"] = df[expenditure_columns].sum(axis=1)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.642738Z","iopub.execute_input":"2022-06-16T02:36:26.643556Z","iopub.status.idle":"2022-06-16T02:36:26.648968Z","shell.execute_reply.started":"2022-06-16T02:36:26.643528Z","shell.execute_reply":"2022-06-16T02:36:26.647186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = from_expenditure_feantures(train_df,expenditure_columns)\ntest_df  = from_expenditure_feantures(test_df,expenditure_columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.654062Z","iopub.execute_input":"2022-06-16T02:36:26.654486Z","iopub.status.idle":"2022-06-16T02:36:26.66734Z","shell.execute_reply.started":"2022-06-16T02:36:26.654453Z","shell.execute_reply":"2022-06-16T02:36:26.666477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Cabinの分割\n- `from_Cabin`によって`Deck`,`Num`,`Side`に分割する","metadata":{}},{"cell_type":"code","source":"def from_cabin(df):\n    df[[\"CabinDeck\",\"CabinNum\",\"CabinSide\"]] = df[\"Cabin\"].str.split(\"/\",expand=True)\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.670413Z","iopub.execute_input":"2022-06-16T02:36:26.670902Z","iopub.status.idle":"2022-06-16T02:36:26.677246Z","shell.execute_reply.started":"2022-06-16T02:36:26.670864Z","shell.execute_reply":"2022-06-16T02:36:26.675533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = from_cabin(train_df)\ntest_df  = from_cabin(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.678757Z","iopub.execute_input":"2022-06-16T02:36:26.679755Z","iopub.status.idle":"2022-06-16T02:36:26.728569Z","shell.execute_reply.started":"2022-06-16T02:36:26.679716Z","shell.execute_reply":"2022-06-16T02:36:26.727374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 最頻値埋め\n`HomePlanet`,`CryoSleep`,`Destination`を最頻値によってnan埋めする","metadata":{}},{"cell_type":"code","source":"def simple_mode_replacement(df,columns):\n    df[columns] = df[columns].fillna(df[columns].mode().iloc[0])\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.729904Z","iopub.execute_input":"2022-06-16T02:36:26.730318Z","iopub.status.idle":"2022-06-16T02:36:26.739573Z","shell.execute_reply.started":"2022-06-16T02:36:26.730265Z","shell.execute_reply":"2022-06-16T02:36:26.737138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"columns = [\"HomePlanet\",\"CryoSleep\",\"Destination\"]\ntrain_df= simple_mode_replacement(train_df,columns)\ntest_df = simple_mode_replacement(test_df,columns)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.741263Z","iopub.execute_input":"2022-06-16T02:36:26.741941Z","iopub.status.idle":"2022-06-16T02:36:26.783339Z","shell.execute_reply.started":"2022-06-16T02:36:26.741894Z","shell.execute_reply":"2022-06-16T02:36:26.781766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cabinのほう\ndef group_mode_replacement(df:pd.DataFrame, groupby:str or list,column:str) -> pd.DataFrame:\n    # Find all passengers belonging to groups where at least one member has a non-null column value\n    # 少なくとも一人，nullではないカラム値を持つメンバーを検索する\n    temp = df.groupby(groupby).filter(lambda x: x[column].notna().any())\n    func = lambda x: x.fillna(x.mode().iloc[0]) if x.isna().any() else x\n    temp[column] = temp.groupby(groupby)[column].transform(func)\n\n    df.loc[temp.index,column] = temp[column]\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.785501Z","iopub.execute_input":"2022-06-16T02:36:26.786221Z","iopub.status.idle":"2022-06-16T02:36:26.795974Z","shell.execute_reply.started":"2022-06-16T02:36:26.786173Z","shell.execute_reply":"2022-06-16T02:36:26.794629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = group_mode_replacement(train_df,groupby=\"GroupId\",column=\"Cabin\")\ntest_df  = group_mode_replacement(test_df,groupby=\"GroupId\",column=\"Cabin\")","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:26.797821Z","iopub.execute_input":"2022-06-16T02:36:26.799583Z","iopub.status.idle":"2022-06-16T02:36:33.658126Z","shell.execute_reply.started":"2022-06-16T02:36:26.799539Z","shell.execute_reply":"2022-06-16T02:36:33.656466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"Cabin\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.660024Z","iopub.execute_input":"2022-06-16T02:36:33.660571Z","iopub.status.idle":"2022-06-16T02:36:33.67057Z","shell.execute_reply.started":"2022-06-16T02:36:33.660523Z","shell.execute_reply":"2022-06-16T02:36:33.669378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"Cabin\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.672171Z","iopub.execute_input":"2022-06-16T02:36:33.672684Z","iopub.status.idle":"2022-06-16T02:36:33.685851Z","shell.execute_reply.started":"2022-06-16T02:36:33.672648Z","shell.execute_reply":"2022-06-16T02:36:33.68498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### まだnanある\nこれだけだとtrainに99のnanがtestに63のnanがあるので`HomePlanet`,`Destination`の最頻値によってnan埋めする","metadata":{}},{"cell_type":"code","source":"train_df = group_mode_replacement(train_df,groupby=[\"HomePlanet\",\"Destination\"],column=\"Cabin\")\ntest_df = group_mode_replacement(test_df,[\"HomePlanet\",\"Destination\"],\"Cabin\")","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.687393Z","iopub.execute_input":"2022-06-16T02:36:33.688505Z","iopub.status.idle":"2022-06-16T02:36:33.770494Z","shell.execute_reply.started":"2022-06-16T02:36:33.688465Z","shell.execute_reply":"2022-06-16T02:36:33.769355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cabinの分割\ntrain_df = from_cabin(train_df)\ntest_df = from_cabin(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.77192Z","iopub.execute_input":"2022-06-16T02:36:33.772261Z","iopub.status.idle":"2022-06-16T02:36:33.81574Z","shell.execute_reply.started":"2022-06-16T02:36:33.772229Z","shell.execute_reply":"2022-06-16T02:36:33.814736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VIPのnan個数確認\ntrain_df[\"VIP\"].isna().sum(),test_df[\"VIP\"].isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.81694Z","iopub.execute_input":"2022-06-16T02:36:33.817278Z","iopub.status.idle":"2022-06-16T02:36:33.826479Z","shell.execute_reply.started":"2022-06-16T02:36:33.817245Z","shell.execute_reply":"2022-06-16T02:36:33.824495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### VIPについて\nVIPはデータより以下のことが言える\n- 支出が0でコールドスリープをしていない乗客はVIPではない\n- 12歳以下の乗客はVIPではない\n- 地球からの乗客はVIPではない\n- 火星からのVIPは18歳以上で，コールドスリープをせず，`5 Cancri e`にはいかない\nこのことより","metadata":{}},{"cell_type":"code","source":"def impute_vip_for_no_spend(df:pd.DataFrame) -> pd.DataFrame:\n    #\"VIP\"がnull & 合計支出が0 & \"CryoSleep\"をしていない\n    df.loc[(df[\"VIP\"].isna()) & (df[\"TotalExpense\"] == 0.0) & (~df[\"CryoSleep\"]),\"VIP\"] = False\n    return df\n\ndef impute_vip_for_children(df:pd.DataFrame) -> pd.DataFrame:\n    df.loc[(df[\"VIP\"].isna()) & (df[\"Age\"] <= 12), \"VIP\"] = False\n    return df\n\ndef impute_vip_for_earthling(df:pd.DataFrame) -> pd.DataFrame:\n    df.loc[(df[\"VIP\"].isna()) & (df[\"HomePlanet\"] == \"Earth\"), \"VIP\"] = False\n    return df\n\ndef impute_vip_for_martians(df:pd.DataFrame) ->pd.DataFrame:\n    df.loc[(df[\"VIP\"].isna()) & (df[\"Age\"] >= 18) & (~df[\"CryoSleep\"]) & (df[\"Destination\"] != \"55 cancri e\"),\"VIP\"] = True\n    return df\n\ndef impute_vip(df:pd.DataFrame) -> pd.DataFrame:\n    df = impute_vip_for_no_spend(df)\n    df = impute_vip_for_children(df)\n    df = impute_vip_for_earthling(df)\n    df = impute_vip_for_martians(df)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.827982Z","iopub.execute_input":"2022-06-16T02:36:33.828468Z","iopub.status.idle":"2022-06-16T02:36:33.839903Z","shell.execute_reply.started":"2022-06-16T02:36:33.828436Z","shell.execute_reply":"2022-06-16T02:36:33.838846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = impute_vip(train_df)\ntest_df  = impute_vip(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.841057Z","iopub.execute_input":"2022-06-16T02:36:33.844215Z","iopub.status.idle":"2022-06-16T02:36:33.865992Z","shell.execute_reply.started":"2022-06-16T02:36:33.844174Z","shell.execute_reply":"2022-06-16T02:36:33.864851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"VIP\"].isna().sum()\n# ちょっと残った","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.868542Z","iopub.execute_input":"2022-06-16T02:36:33.869032Z","iopub.status.idle":"2022-06-16T02:36:33.883027Z","shell.execute_reply.started":"2022-06-16T02:36:33.868997Z","shell.execute_reply":"2022-06-16T02:36:33.881432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ランダムで決める\ndef impute_vip_by_prob(df):\n    probs = df[\"VIP\"].value_counts() / df[\"VIP\"].notna().sum()\n    values = np.random.choice([False, True], size=df[\"VIP\"].isna().sum(), p=probs)\n    df.loc[df[\"VIP\"].isna(), \"VIP\"] = values\n    df[\"VIP\"] = df[\"VIP\"].astype(bool)\n    return df\n\ntrain_df = impute_vip_by_prob(train_df)\ntest_df = impute_vip_by_prob(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.884578Z","iopub.execute_input":"2022-06-16T02:36:33.885386Z","iopub.status.idle":"2022-06-16T02:36:33.90881Z","shell.execute_reply.started":"2022-06-16T02:36:33.885336Z","shell.execute_reply":"2022-06-16T02:36:33.907929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"VIP\"].isna().sum()\n# 埋められた","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.910006Z","iopub.execute_input":"2022-06-16T02:36:33.91067Z","iopub.status.idle":"2022-06-16T02:36:33.918977Z","shell.execute_reply.started":"2022-06-16T02:36:33.910637Z","shell.execute_reply":"2022-06-16T02:36:33.917931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### いらない特徴量の削除\n`PassengerId`,`Cabin`,`Name`は到着したかどうかに関係しないと考え削除","metadata":{}},{"cell_type":"code","source":"drop = [\"PassengerId\",\"Cabin\",\"Name\"]\ntrain_df = train_df.drop(drop,axis=1)\ntest_df  = test_df.drop(drop,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.920444Z","iopub.execute_input":"2022-06-16T02:36:33.921388Z","iopub.status.idle":"2022-06-16T02:36:33.932692Z","shell.execute_reply.started":"2022-06-16T02:36:33.921347Z","shell.execute_reply":"2022-06-16T02:36:33.931893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nullの確認\ntrain_df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.940233Z","iopub.execute_input":"2022-06-16T02:36:33.940867Z","iopub.status.idle":"2022-06-16T02:36:33.957332Z","shell.execute_reply.started":"2022-06-16T02:36:33.940818Z","shell.execute_reply":"2022-06-16T02:36:33.955745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 特徴量のエンコーディング\n二つのデータセットをつなげる`concat_train_test()`と分ける`split_train_test()`  \nでも`CabinNum`と`GroupSize`はexperimentの一つなので無視します．","metadata":{}},{"cell_type":"code","source":"def concat_train_test(train:pd.DataFrame,test:pd.DataFrame,has_labels=False) -> tuple:\n    transported = None\n    \n    #testデータにこのラベルはないのでもしもあったら削除しなきゃ\n    if has_labels is True:\n        transported = train[\"Transported\"].copy()\n        train = train.drop(\"Transported\",axis=1)\n    \n    train_index = train.index\n    test_index  = test.index\n\n    df = pd.concat([train,test])\n\n    return df,train_index,test_index,transported\n\ndef split_train_test(df,train_index,test_index,transported=None):\n    train_df = df.loc[train_index,:]\n    if transported is not None:\n        train_df[\"Transported\"] = transported\n    \n    test_df = df.loc[test_index,:]\n\n    return train_df,test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.960465Z","iopub.execute_input":"2022-06-16T02:36:33.961059Z","iopub.status.idle":"2022-06-16T02:36:33.973146Z","shell.execute_reply.started":"2022-06-16T02:36:33.961012Z","shell.execute_reply":"2022-06-16T02:36:33.97134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df, train_idx, test_idx, transported = concat_train_test(train_df, test_df, has_labels=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:33.975659Z","iopub.execute_input":"2022-06-16T02:36:33.976417Z","iopub.status.idle":"2022-06-16T02:36:34.027041Z","shell.execute_reply.started":"2022-06-16T02:36:33.976356Z","shell.execute_reply":"2022-06-16T02:36:34.026186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### bool2int\nロジスティック回帰ではカテゴリ変数をboolのまま操作できないのでintに変換します","metadata":{}},{"cell_type":"code","source":"def bool2int(df):\n    columns = [column for column in df.columns if df[column].dtype.name == \"bool\"]\n    df[columns] = df[columns].astype(int)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.028547Z","iopub.execute_input":"2022-06-16T02:36:34.029061Z","iopub.status.idle":"2022-06-16T02:36:34.035252Z","shell.execute_reply.started":"2022-06-16T02:36:34.029022Z","shell.execute_reply":"2022-06-16T02:36:34.034268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = bool2int(df)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.036893Z","iopub.execute_input":"2022-06-16T02:36:34.037438Z","iopub.status.idle":"2022-06-16T02:36:34.057908Z","shell.execute_reply.started":"2022-06-16T02:36:34.037397Z","shell.execute_reply":"2022-06-16T02:36:34.056733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### CabinSide\nCabinSideは`S`と`P`しかないのでバイナリに変換できます","metadata":{}},{"cell_type":"code","source":"df[\"CabinSide\"] = df[\"CabinSide\"].map({\"S\":0,\"P\":1})","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.059718Z","iopub.execute_input":"2022-06-16T02:36:34.060066Z","iopub.status.idle":"2022-06-16T02:36:34.068736Z","shell.execute_reply.started":"2022-06-16T02:36:34.060035Z","shell.execute_reply":"2022-06-16T02:36:34.067624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### カテゴリ変数のダミー化","metadata":{}},{"cell_type":"code","source":"to_be_encoded = [\"HomePlanet\",\"Destination\",\"GroupSize\",\"CabinDeck\"]\ndf = pd.get_dummies(df,columns=to_be_encoded)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.070926Z","iopub.execute_input":"2022-06-16T02:36:34.071505Z","iopub.status.idle":"2022-06-16T02:36:34.107343Z","shell.execute_reply.started":"2022-06-16T02:36:34.071455Z","shell.execute_reply":"2022-06-16T02:36:34.10624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, test_df = split_train_test(df, train_idx, test_idx, transported=transported)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.108954Z","iopub.execute_input":"2022-06-16T02:36:34.109365Z","iopub.status.idle":"2022-06-16T02:36:34.131871Z","shell.execute_reply.started":"2022-06-16T02:36:34.109328Z","shell.execute_reply":"2022-06-16T02:36:34.13081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### KNN\n欠損値はKNNをしようして埋める\n- `GroupId`と`cabinNum`に対してエンコードをする","metadata":{}},{"cell_type":"code","source":"def impute_missing_using_knn(df, numeric_cols, has_labels=False):\n    x = df\n    \n    if has_labels is True:\n        transported = df[\"Transported\"]\n        x = df.drop(\"Transported\", axis=1)\n        \n    scaler = preprocessing.StandardScaler()\n    x[numeric_cols] = scaler.fit_transform(x[numeric_cols])\n    \n    imputer = impute.KNNImputer(n_neighbors=5, weights=\"distance\")\n    x = imputer.fit_transform(x)\n    \n    if has_labels is True:\n        x = np.hstack((x, transported.values.reshape(-1, 1)))\n        \n    return pd.DataFrame(x, columns=df.columns, index=df.index)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.133954Z","iopub.execute_input":"2022-06-16T02:36:34.134614Z","iopub.status.idle":"2022-06-16T02:36:34.14387Z","shell.execute_reply.started":"2022-06-16T02:36:34.134574Z","shell.execute_reply":"2022-06-16T02:36:34.142951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cabin_num = train_df[\"CabinNum\"]\ntrain_group_id  = train_df[\"GroupId\"]\n\ntest_cabin_num  = test_df[\"CabinNum\"]\ntest_group_df   = test_df[\"GroupId\"]\n\nto_drop = [\"GroupId\", \"CabinNum\"]\nnumeric_cols = [\"Age\", \"TotalExpense\"] + expenditure_columns\n\ntrain_df = impute_missing_using_knn(train_df.drop(to_drop, axis=1), numeric_cols, has_labels=True)\ntest_df = impute_missing_using_knn(test_df.drop(to_drop, axis=1), numeric_cols)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:34.145746Z","iopub.execute_input":"2022-06-16T02:36:34.146506Z","iopub.status.idle":"2022-06-16T02:36:35.261476Z","shell.execute_reply.started":"2022-06-16T02:36:34.146454Z","shell.execute_reply":"2022-06-16T02:36:35.259756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 確認\ntrain_df.isna().any()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.263041Z","iopub.execute_input":"2022-06-16T02:36:35.263452Z","iopub.status.idle":"2022-06-16T02:36:35.274687Z","shell.execute_reply.started":"2022-06-16T02:36:35.263416Z","shell.execute_reply":"2022-06-16T02:36:35.273715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### クロスバリデーションのためにkfoldを設定","metadata":{}},{"cell_type":"code","source":"train_df = train_df.reset_index()\n\n# kfoldカラムの追加\ntrain_df[\"kfold\"] = -1\nkf = model_selection.KFold(n_splits=5,random_state=42,shuffle=True)\n\nfor idx, (_,val_idx) in enumerate(kf.split(train_df)):\n    train_df.loc[val_idx,\"kfold\"] = idx\n\n#PassengerIdをIndexにする\ntrain_df = train_df.set_index(\"PassengerId\")\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.276197Z","iopub.execute_input":"2022-06-16T02:36:35.277212Z","iopub.status.idle":"2022-06-16T02:36:35.324236Z","shell.execute_reply.started":"2022-06-16T02:36:35.277176Z","shell.execute_reply":"2022-06-16T02:36:35.322804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### データクレンジングおわり\n#### csvとして保存しよう","metadata":{}},{"cell_type":"code","source":"train_df.to_csv(\"train_prepared.csv\",index=False)\ntest_df.to_csv(\"test_prepared.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.325811Z","iopub.execute_input":"2022-06-16T02:36:35.326879Z","iopub.status.idle":"2022-06-16T02:36:35.749714Z","shell.execute_reply.started":"2022-06-16T02:36:35.326823Z","shell.execute_reply":"2022-06-16T02:36:35.748268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"train_prepared.csv\")\ntest_df = pd.read_csv(\"test_prepared.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.751495Z","iopub.execute_input":"2022-06-16T02:36:35.751945Z","iopub.status.idle":"2022-06-16T02:36:35.842304Z","shell.execute_reply.started":"2022-06-16T02:36:35.751891Z","shell.execute_reply":"2022-06-16T02:36:35.841189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ロジスティック回帰","metadata":{}},{"cell_type":"code","source":"from sklearn import linear_model,metrics\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.844843Z","iopub.execute_input":"2022-06-16T02:36:35.845549Z","iopub.status.idle":"2022-06-16T02:36:35.945926Z","shell.execute_reply.started":"2022-06-16T02:36:35.845493Z","shell.execute_reply":"2022-06-16T02:36:35.944371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 交差検証\n5回に分けて交差検証を行う\n- 今回はロジスティック回帰とランダムフォレストで実行した結果ロジスティック回帰のほうが精度がよかった．","metadata":{}},{"cell_type":"code","source":"def train(df):\n    df[\"preds\"] = pd.NA\n\n    drop = [\"Transported\",\"preds\",\"kfold\"]\n\n    for fold in range(5):\n        train = df[df[\"kfold\"] != fold]\n\n        y_train = train[\"Transported\"].values\n        x_train = train.drop(drop,axis=1).values\n\n        val = df[df[\"kfold\"] == fold]\n\n        y_val = val[\"Transported\"].values\n        x_val = val.drop(drop,axis=1).values\n\n        # model = RandomForestClassifier(max_depth=10,max_features=\"auto\",n_estimators=100)\n        model = linear_model.LogisticRegression(max_iter=1000)\n        model.fit(x_train,y_train)\n\n        preds = model.predict(x_val)\n        df.loc[val.index,\"preds\"] = preds\n\n        acc = metrics.accuracy_score(y_val,preds)\n        print(f\"Fold{fold+1}-Accuracy = {acc:.4f}\")\n    \n    df[drop] = df[drop].astype(int)\n\n    # 全体の正解率計算\n    acc = metrics.accuracy_score(df[\"Transported\"].values,df[\"preds\"].values)\n    print(f\"Overall accuracy:{acc:.4f}\")\n\n    return df,model","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.947807Z","iopub.execute_input":"2022-06-16T02:36:35.948438Z","iopub.status.idle":"2022-06-16T02:36:35.962445Z","shell.execute_reply.started":"2022-06-16T02:36:35.948369Z","shell.execute_reply":"2022-06-16T02:36:35.961381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_id = test_df[\"PassengerId\"]\ndel test_df[\"PassengerId\"]","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.964069Z","iopub.execute_input":"2022-06-16T02:36:35.964516Z","iopub.status.idle":"2022-06-16T02:36:35.981235Z","shell.execute_reply.started":"2022-06-16T02:36:35.964474Z","shell.execute_reply":"2022-06-16T02:36:35.98012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_values = test_df.values","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.98301Z","iopub.execute_input":"2022-06-16T02:36:35.983466Z","iopub.status.idle":"2022-06-16T02:36:35.995092Z","shell.execute_reply.started":"2022-06-16T02:36:35.983415Z","shell.execute_reply":"2022-06-16T02:36:35.993729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_exp1,model = train(train_df.copy())","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:35.998121Z","iopub.execute_input":"2022-06-16T02:36:35.998701Z","iopub.status.idle":"2022-06-16T02:36:36.603455Z","shell.execute_reply.started":"2022-06-16T02:36:35.998647Z","shell.execute_reply":"2022-06-16T02:36:36.602475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_df_values)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:36.60487Z","iopub.execute_input":"2022-06-16T02:36:36.606188Z","iopub.status.idle":"2022-06-16T02:36:36.611737Z","shell.execute_reply.started":"2022-06-16T02:36:36.606143Z","shell.execute_reply":"2022-06-16T02:36:36.610876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ans = [\"PassengerId\",\"Transported\"]\n\nans_df = pd.DataFrame()\nans_df[\"PassengerId\"] = test_id\nans_df[\"Transported\"] = pred.astype(bool)\nans_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:36.613433Z","iopub.execute_input":"2022-06-16T02:36:36.61405Z","iopub.status.idle":"2022-06-16T02:36:36.63681Z","shell.execute_reply.started":"2022-06-16T02:36:36.614009Z","shell.execute_reply":"2022-06-16T02:36:36.635942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"today = date.today()\nmonth = today.month\nday   = today.day\n\ncount = 2\nans_df.to_csv(f\"submit_{str(month)}{str(day)}_{str(count)}.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T02:36:36.638344Z","iopub.execute_input":"2022-06-16T02:36:36.638862Z","iopub.status.idle":"2022-06-16T02:36:36.650186Z","shell.execute_reply.started":"2022-06-16T02:36:36.638821Z","shell.execute_reply":"2022-06-16T02:36:36.64922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}