{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## **Table of contents**\n1. Imports\n1. Data Loading\n1. Data Cleaning\n1. Model training","metadata":{}},{"cell_type":"markdown","source":"## **Imports**","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-19T17:50:40.513889Z","iopub.execute_input":"2022-07-19T17:50:40.514949Z","iopub.status.idle":"2022-07-19T17:50:40.537404Z","shell.execute_reply.started":"2022-07-19T17:50:40.514809Z","shell.execute_reply":"2022-07-19T17:50:40.535987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import clear_output\nclear_output()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:40.539738Z","iopub.execute_input":"2022-07-19T17:50:40.540625Z","iopub.status.idle":"2022-07-19T17:50:40.552399Z","shell.execute_reply.started":"2022-07-19T17:50:40.540525Z","shell.execute_reply":"2022-07-19T17:50:40.551064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data analysis and wrangling\nimport pandas as pd\nimport numpy as np\n# import random as rnd\n\n# visualization\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# machine learning\nfrom lightgbm import LGBMClassifier\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import cross_val_score\n\n# progress bar display\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:40.553994Z","iopub.execute_input":"2022-07-19T17:50:40.554511Z","iopub.status.idle":"2022-07-19T17:50:41.174678Z","shell.execute_reply.started":"2022-07-19T17:50:40.554462Z","shell.execute_reply":"2022-07-19T17:50:41.173322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Loading**","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('../input/spaceship-titanic/train.csv')\ntest_df = pd.read_csv('../input/spaceship-titanic/test.csv')\ntest_ids = test_df['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.179129Z","iopub.execute_input":"2022-07-19T17:50:41.179520Z","iopub.status.idle":"2022-07-19T17:50:41.239050Z","shell.execute_reply.started":"2022-07-19T17:50:41.179485Z","shell.execute_reply":"2022-07-19T17:50:41.237701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Data Cleaning**","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.240622Z","iopub.execute_input":"2022-07-19T17:50:41.241448Z","iopub.status.idle":"2022-07-19T17:50:41.268505Z","shell.execute_reply.started":"2022-07-19T17:50:41.241406Z","shell.execute_reply":"2022-07-19T17:50:41.267592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.270259Z","iopub.execute_input":"2022-07-19T17:50:41.270902Z","iopub.status.idle":"2022-07-19T17:50:41.307820Z","shell.execute_reply.started":"2022-07-19T17:50:41.270862Z","shell.execute_reply":"2022-07-19T17:50:41.306559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.309206Z","iopub.execute_input":"2022-07-19T17:50:41.309585Z","iopub.status.idle":"2022-07-19T17:50:41.365126Z","shell.execute_reply.started":"2022-07-19T17:50:41.309543Z","shell.execute_reply":"2022-07-19T17:50:41.363949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = train_df.drop([\"Transported\"], axis=1)\ny_train = train_df['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.366769Z","iopub.execute_input":"2022-07-19T17:50:41.367125Z","iopub.status.idle":"2022-07-19T17:50:41.374724Z","shell.execute_reply.started":"2022-07-19T17:50:41.367092Z","shell.execute_reply":"2022-07-19T17:50:41.373791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data = pd.concat([X_train, test_df], axis=0)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.376113Z","iopub.execute_input":"2022-07-19T17:50:41.376666Z","iopub.status.idle":"2022-07-19T17:50:41.392216Z","shell.execute_reply.started":"2022-07-19T17:50:41.376627Z","shell.execute_reply":"2022-07-19T17:50:41.391124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"僅一小部分的資料缺失，整個column或整個row丟掉太可惜，因此我們試著用填的","metadata":{}},{"cell_type":"code","source":"missing_values_count = concat_data.isnull().sum()\nmissing_values_count","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.393834Z","iopub.execute_input":"2022-07-19T17:50:41.394652Z","iopub.status.idle":"2022-07-19T17:50:41.418195Z","shell.execute_reply.started":"2022-07-19T17:50:41.394602Z","shell.execute_reply":"2022-07-19T17:50:41.417320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"一起訂票 -> 同一個 Cabin -> 相鄰的 rows 的 Cabin 高機率相同 -> 用 bfill 來 impute","metadata":{}},{"cell_type":"code","source":"concat_data.loc[concat_data['Cabin'] == 'G/734/S']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.419493Z","iopub.execute_input":"2022-07-19T17:50:41.420499Z","iopub.status.idle":"2022-07-19T17:50:41.456907Z","shell.execute_reply.started":"2022-07-19T17:50:41.420456Z","shell.execute_reply":"2022-07-19T17:50:41.455503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data['Cabin'] = concat_data['Cabin'].fillna(method = 'bfill')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.458586Z","iopub.execute_input":"2022-07-19T17:50:41.459078Z","iopub.status.idle":"2022-07-19T17:50:41.468246Z","shell.execute_reply.started":"2022-07-19T17:50:41.459037Z","shell.execute_reply":"2022-07-19T17:50:41.466827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"由於找不到 Age 與其他 features 間明顯的關係，我們以平均值來 impute Age","metadata":{}},{"cell_type":"code","source":"concat_data['Age'].fillna(concat_data['Age'].mean(), inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.476493Z","iopub.execute_input":"2022-07-19T17:50:41.476914Z","iopub.status.idle":"2022-07-19T17:50:41.484328Z","shell.execute_reply.started":"2022-07-19T17:50:41.476880Z","shell.execute_reply":"2022-07-19T17:50:41.482997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"非 VIP 遠大於 VIP 因此以 False 來 impute VIP  \n要去 TRAPPIST-1e 的人遠大於其他，因此以 TRAPPIST-1e 來 impute Destination\n","metadata":{}},{"cell_type":"code","source":"concat_data.groupby('VIP').count()['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.486467Z","iopub.execute_input":"2022-07-19T17:50:41.487302Z","iopub.status.idle":"2022-07-19T17:50:41.517403Z","shell.execute_reply.started":"2022-07-19T17:50:41.487250Z","shell.execute_reply":"2022-07-19T17:50:41.516617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data.groupby(['Destination']).count()['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.518667Z","iopub.execute_input":"2022-07-19T17:50:41.519155Z","iopub.status.idle":"2022-07-19T17:50:41.542157Z","shell.execute_reply.started":"2022-07-19T17:50:41.519124Z","shell.execute_reply":"2022-07-19T17:50:41.541053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data.VIP.fillna(False, inplace=True)\nconcat_data.Destination.fillna('TRAPPIST-1e', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.543478Z","iopub.execute_input":"2022-07-19T17:50:41.544283Z","iopub.status.idle":"2022-07-19T17:50:41.554627Z","shell.execute_reply.started":"2022-07-19T17:50:41.544246Z","shell.execute_reply":"2022-07-19T17:50:41.553749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"VIP 跟來自哪裡有關，若是VIP，則最有可能來自 Europa ，若不是 VIP 則最有可能來自地球","metadata":{}},{"cell_type":"code","source":"print('VIPs are from:')\nconcat_data.loc[concat_data['VIP']==True].groupby(['HomePlanet']).count().sort_values(by='VIP', ascending=False)['VIP']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.555704Z","iopub.execute_input":"2022-07-19T17:50:41.556639Z","iopub.status.idle":"2022-07-19T17:50:41.575843Z","shell.execute_reply.started":"2022-07-19T17:50:41.556600Z","shell.execute_reply":"2022-07-19T17:50:41.574500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('others are from:')\nconcat_data.loc[concat_data['VIP']==False].groupby(['HomePlanet']).count().sort_values(by='VIP', ascending=False)['VIP']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.577559Z","iopub.execute_input":"2022-07-19T17:50:41.578027Z","iopub.status.idle":"2022-07-19T17:50:41.606614Z","shell.execute_reply.started":"2022-07-19T17:50:41.577992Z","shell.execute_reply":"2022-07-19T17:50:41.605400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data.loc[(concat_data['VIP'] == True),'HomePlanet'] = concat_data.loc[(concat_data['VIP'] == True),'HomePlanet'].fillna('Europa')\nconcat_data.loc[(concat_data['VIP'] == False),'HomePlanet'] = concat_data.loc[(concat_data['VIP'] == False),'HomePlanet'].fillna('Earth')","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.608270Z","iopub.execute_input":"2022-07-19T17:50:41.608763Z","iopub.status.idle":"2022-07-19T17:50:41.623954Z","shell.execute_reply.started":"2022-07-19T17:50:41.608726Z","shell.execute_reply":"2022-07-19T17:50:41.622853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data['total_bill'] = (concat_data['RoomService'] + concat_data['FoodCourt'] + concat_data['ShoppingMall'] + concat_data['Spa'] + concat_data['VRDeck']).fillna(0).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.625779Z","iopub.execute_input":"2022-07-19T17:50:41.626762Z","iopub.status.idle":"2022-07-19T17:50:41.634798Z","shell.execute_reply.started":"2022-07-19T17:50:41.626723Z","shell.execute_reply":"2022-07-19T17:50:41.633646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"若有買東西，則不可能處於低溫睡眠狀態，若沒買東西，則有較高機率處於低溫睡眠狀態  \n因此跟據 total_bill 來決定是否處於低溫狀態","metadata":{}},{"cell_type":"code","source":"concat_data.loc[concat_data['total_bill']==0].groupby('CryoSleep').count()['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.636471Z","iopub.execute_input":"2022-07-19T17:50:41.637504Z","iopub.status.idle":"2022-07-19T17:50:41.658704Z","shell.execute_reply.started":"2022-07-19T17:50:41.637467Z","shell.execute_reply":"2022-07-19T17:50:41.657300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data.loc[concat_data['total_bill']!=0].groupby('CryoSleep').count()['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.660109Z","iopub.execute_input":"2022-07-19T17:50:41.660430Z","iopub.status.idle":"2022-07-19T17:50:41.683222Z","shell.execute_reply.started":"2022-07-19T17:50:41.660402Z","shell.execute_reply":"2022-07-19T17:50:41.682041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 若無消費 -> CryoSleep = True\nconcat_data.loc[(concat_data['total_bill'] == 0),'CryoSleep'] = concat_data.loc[(concat_data['total_bill'] == 0),'CryoSleep'].fillna(True)\n# 若有消費 -> CryoSleep = False\nconcat_data.loc[(concat_data['total_bill'] != 0),'CryoSleep'] = concat_data.loc[(concat_data['total_bill'] != 0),'CryoSleep'].fillna(False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.685155Z","iopub.execute_input":"2022-07-19T17:50:41.685517Z","iopub.status.idle":"2022-07-19T17:50:41.702285Z","shell.execute_reply.started":"2022-07-19T17:50:41.685483Z","shell.execute_reply":"2022-07-19T17:50:41.700849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"消費能力與是否進入低溫睡眠以及是否為 VIP 皆有關","metadata":{}},{"cell_type":"code","source":"concat_data.groupby(['VIP']).mean().sort_values(by='total_bill', ascending=False)['total_bill']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.703790Z","iopub.execute_input":"2022-07-19T17:50:41.704642Z","iopub.status.idle":"2022-07-19T17:50:41.718919Z","shell.execute_reply.started":"2022-07-19T17:50:41.704592Z","shell.execute_reply":"2022-07-19T17:50:41.717683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols = ['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n# 若 CryoSleep = True -> 不會使用這些服務 -> 用 0 來 impute\nfor col in cols:\n    concat_data.loc[(concat_data['CryoSleep'] == 1),col] = concat_data.loc[(concat_data['CryoSleep'] == 1),col].fillna(0)\n\n# 若 CryoSleep = False, 則用平均數來 impute (由於 VIP 普遍有較高的消費能力，因此 VIP 的 missing value 以 VIP 的平均來 impute， non VIP 的)\nm1 = (concat_data['CryoSleep'] == 0) & (concat_data['VIP'] == 0)\nm2 = (concat_data['CryoSleep'] == 0) & (concat_data['VIP'] == 1)\nfor col in cols:\n    concat_data.loc[m1,col] = concat_data.loc[m1,col].fillna(concat_data.loc[m1,col].mean())\n    concat_data.loc[m2,col] = concat_data.loc[m2,col].fillna(concat_data.loc[m2,col].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.721007Z","iopub.execute_input":"2022-07-19T17:50:41.721501Z","iopub.status.idle":"2022-07-19T17:50:41.784334Z","shell.execute_reply.started":"2022-07-19T17:50:41.721466Z","shell.execute_reply":"2022-07-19T17:50:41.783101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"在把'PassengerId', 'Cabin', 'Name' 丟掉之前，最後確認一下他們是不是真的沒用  \n'PassengerId' 由 'group' 和 'num_in_group' 二個部分組成  \n'Cabin' 由 'deck', 'num' 和 'side' 三個部分組成  \n'Name' 由 'first_name' 和 'last_name' 二個部分組成\n'num_in_group', 'deck' 和 'side' 似乎與 'Transported' 有關  \n其餘 'PassengerId', 'group', 'Cabin', 'num', 'Name', 'first_name' 和 'last_name' 捨去","metadata":{}},{"cell_type":"code","source":"temp = train_df.copy()\ntemp['Transported'] = temp['Transported'].astype(int)\ntemp[['group', 'num_in_group']] = train_df['PassengerId'].str.split('_', expand=True)\ntemp[['deck', 'num','side']] = temp['Cabin'].str.split('/', expand=True)\ntemp[['first_name', 'last_name']] = train_df['Name'].str.split(' ', expand=True)\n\nconcat_data[['group', 'num_in_group']] = concat_data['PassengerId'].str.split('_', expand=True)\nconcat_data[['deck', 'num','side']] = concat_data['Cabin'].str.split('/', expand=True)\nconcat_data[['first_name', 'last_name']] = concat_data['Name'].str.split(' ', expand=True)\n\nconcat_data['deck'] = concat_data['deck'].map( {'T': 0, 'E': 1, 'D': 2, 'F': 3, 'A': 4, 'G': 5, 'B': 6, 'C': 7} ).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.801433Z","iopub.execute_input":"2022-07-19T17:50:41.802231Z","iopub.status.idle":"2022-07-19T17:50:41.942000Z","shell.execute_reply.started":"2022-07-19T17:50:41.802182Z","shell.execute_reply":"2022-07-19T17:50:41.941090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:41.943253Z","iopub.execute_input":"2022-07-19T17:50:41.944488Z","iopub.status.idle":"2022-07-19T17:50:42.037598Z","shell.execute_reply.started":"2022-07-19T17:50:41.944446Z","shell.execute_reply":"2022-07-19T17:50:42.036632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['deck','Transported']].groupby(['deck']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.038815Z","iopub.execute_input":"2022-07-19T17:50:42.039881Z","iopub.status.idle":"2022-07-19T17:50:42.055815Z","shell.execute_reply.started":"2022-07-19T17:50:42.039831Z","shell.execute_reply":"2022-07-19T17:50:42.054393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['side','Transported']].groupby(['side']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.057515Z","iopub.execute_input":"2022-07-19T17:50:42.058673Z","iopub.status.idle":"2022-07-19T17:50:42.075102Z","shell.execute_reply.started":"2022-07-19T17:50:42.058629Z","shell.execute_reply":"2022-07-19T17:50:42.074101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['HomePlanet','Transported']].groupby(['HomePlanet']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.076562Z","iopub.execute_input":"2022-07-19T17:50:42.077513Z","iopub.status.idle":"2022-07-19T17:50:42.090785Z","shell.execute_reply.started":"2022-07-19T17:50:42.077476Z","shell.execute_reply":"2022-07-19T17:50:42.089781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['CryoSleep','Transported']].groupby(['CryoSleep']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.092375Z","iopub.execute_input":"2022-07-19T17:50:42.093052Z","iopub.status.idle":"2022-07-19T17:50:42.106157Z","shell.execute_reply.started":"2022-07-19T17:50:42.093002Z","shell.execute_reply":"2022-07-19T17:50:42.104879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['Destination','Transported']].groupby(['Destination']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.107730Z","iopub.execute_input":"2022-07-19T17:50:42.108350Z","iopub.status.idle":"2022-07-19T17:50:42.126219Z","shell.execute_reply.started":"2022-07-19T17:50:42.108313Z","shell.execute_reply":"2022-07-19T17:50:42.124949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[['VIP','Transported']].groupby(['VIP']).mean().sort_values(by='Transported', ascending=False)['Transported']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.127614Z","iopub.execute_input":"2022-07-19T17:50:42.128395Z","iopub.status.idle":"2022-07-19T17:50:42.142724Z","shell.execute_reply.started":"2022-07-19T17:50:42.128355Z","shell.execute_reply":"2022-07-19T17:50:42.141279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"concat_data = concat_data.drop(['PassengerId','Cabin','Name','group','num','first_name','last_name','num_in_group'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.144115Z","iopub.execute_input":"2022-07-19T17:50:42.144708Z","iopub.status.idle":"2022-07-19T17:50:42.163494Z","shell.execute_reply.started":"2022-07-19T17:50:42.144668Z","shell.execute_reply":"2022-07-19T17:50:42.162157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"我們已經成功去除掉所有 missing values 了","metadata":{}},{"cell_type":"code","source":"missing_values_count = concat_data.isnull().sum()\nmissing_values_count","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.165436Z","iopub.execute_input":"2022-07-19T17:50:42.166318Z","iopub.status.idle":"2022-07-19T17:50:42.184887Z","shell.execute_reply.started":"2022-07-19T17:50:42.166263Z","shell.execute_reply":"2022-07-19T17:50:42.183619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_data = pd.get_dummies(concat_data)\n\ndummy_train_data = dummy_data.iloc[:X_train.shape[0], :]\ndummy_test_data = dummy_data.iloc[X_train.shape[0]:, :]","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.186495Z","iopub.execute_input":"2022-07-19T17:50:42.186924Z","iopub.status.idle":"2022-07-19T17:50:42.213108Z","shell.execute_reply.started":"2022-07-19T17:50:42.186892Z","shell.execute_reply":"2022-07-19T17:50:42.211783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.215172Z","iopub.execute_input":"2022-07-19T17:50:42.215709Z","iopub.status.idle":"2022-07-19T17:50:42.243327Z","shell.execute_reply.started":"2022-07-19T17:50:42.215655Z","shell.execute_reply":"2022-07-19T17:50:42.241764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler = StandardScaler()\nsc_x = scaler.fit_transform(dummy_train_data)\nsc_test_data_x = scaler.transform(dummy_test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.245039Z","iopub.execute_input":"2022-07-19T17:50:42.246573Z","iopub.status.idle":"2022-07-19T17:50:42.321006Z","shell.execute_reply.started":"2022-07-19T17:50:42.246492Z","shell.execute_reply":"2022-07-19T17:50:42.320048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"train data shape: {sc_x.shape}\")\nprint(f\"test data shape: {sc_test_data_x.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.329517Z","iopub.execute_input":"2022-07-19T17:50:42.330183Z","iopub.status.idle":"2022-07-19T17:50:42.337334Z","shell.execute_reply.started":"2022-07-19T17:50:42.330133Z","shell.execute_reply":"2022-07-19T17:50:42.335864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Model training**","metadata":{}},{"cell_type":"code","source":"estimators = []\nfor i in tqdm(range(50,150,5)):\n    model = LGBMClassifier(n_estimators=i, random_state=1)\n    scores = cross_val_score(model, sc_x, y_train, cv=5)\n    estimators.append(scores.mean())\nclear_output()\nestimators","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:50:42.339311Z","iopub.execute_input":"2022-07-19T17:50:42.339830Z","iopub.status.idle":"2022-07-19T17:51:04.806652Z","shell.execute_reply.started":"2022-07-19T17:50:42.339781Z","shell.execute_reply":"2022-07-19T17:51:04.805596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_score = max(estimators)\nbest_score","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:51:04.808057Z","iopub.execute_input":"2022-07-19T17:51:04.808755Z","iopub.status.idle":"2022-07-19T17:51:04.816907Z","shell.execute_reply.started":"2022-07-19T17:51:04.808715Z","shell.execute_reply":"2022-07-19T17:51:04.815611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_estimators = np.argmax(estimators)*5 + 50\nbest_estimators","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:51:04.818878Z","iopub.execute_input":"2022-07-19T17:51:04.819694Z","iopub.status.idle":"2022-07-19T17:51:04.835518Z","shell.execute_reply.started":"2022-07-19T17:51:04.819641Z","shell.execute_reply":"2022-07-19T17:51:04.833851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = LGBMClassifier(n_estimators = best_estimators, random_state=1)\nfinal_model.fit(sc_x, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:55:19.910288Z","iopub.execute_input":"2022-07-19T17:55:19.910973Z","iopub.status.idle":"2022-07-19T17:55:20.073223Z","shell.execute_reply.started":"2022-07-19T17:55:19.910921Z","shell.execute_reply":"2022-07-19T17:55:20.072218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = final_model.predict(sc_test_data_x)\n\nsubmission_data = pd.DataFrame({\"PassengerId\": test_ids.values, \"Transported\": submission })\n\nsubmission_data.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T17:55:22.821389Z","iopub.execute_input":"2022-07-19T17:55:22.821864Z","iopub.status.idle":"2022-07-19T17:55:22.849642Z","shell.execute_reply.started":"2022-07-19T17:55:22.821828Z","shell.execute_reply":"2022-07-19T17:55:22.848592Z"},"trusted":true},"execution_count":null,"outputs":[]}]}