{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-06T15:54:06.63511Z","iopub.execute_input":"2022-07-06T15:54:06.63568Z","iopub.status.idle":"2022-07-06T15:54:06.647529Z","shell.execute_reply.started":"2022-07-06T15:54:06.635638Z","shell.execute_reply":"2022-07-06T15:54:06.645883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#事前情報ありmap\ndef vip(data):\n    data['VIP']=data['VIP'].map({True:1,False:2})\n    return data\n\n#事前情報なしmap\ndef HP(data):\n    list_HP=data['HomePlanet'].to_list()\n    HP_set=set(list_HP)\n    nu_list=[]\n    HP_list=[]\n\n    for i in range(len(HP_set)):\n        nu_list.append(i)\n        HP_list.append(list(HP_set)[i])\n    dic={}\n    for i in range(len(HP_list)):\n        dic[HP_list[i]]=nu_list[i]\n        \n    data['HomePlanet']=data['HomePlanet'].map(dic)\n    return data\n\ndef cabin(data):\n    cabin_list=data['Cabin'].to_list()\n    cabin_F=[]\n    cabin_num=[]\n    cabin_l=[]\n    for i in range(len(cabin_list)):\n        try:\n            cabin_split=cabin_list[i].split('/')\n            cabin_F.append(cabin_split[0])\n            cabin_num.append(cabin_split[1])\n            cabin_l.append(cabin_split[2])            \n        except:\n            cabin_F.append(9999)\n            cabin_num.append(9999)\n            cabin_l.append(9999)\n\n            \n    data['cabin_first']=cabin_F\n    data['cabin_number']=cabin_num\n    data['cabin_last']=cabin_l\n    data=data.drop(columns=['Cabin'])\n    return data\n\ndef CryoSleep(data):\n    data['CryoSleep']=data['CryoSleep'].map({True:1,False:2})\n    return data\n\ndef Transported(data):\n    data['Transported']=data['Transported'].map({True:0,False:1})\n    return data\n\ndef cabin_first(data):\n    list_cabin_first=data['cabin_first'].to_list()\n    cabin_first_set=set(list_cabin_first)\n    nu_list=[]\n    cabin_first_list=[]\n    \n    for i in range(len(cabin_first_set)):\n        nu_list.append(i)\n        cabin_first_list.append(list(cabin_first_set)[i])\n    dic={}\n    for i in range(len(cabin_first_list)):\n        dic[cabin_first_list[i]]=nu_list[i]\n    \n    data['cabin_first']=data['cabin_first'].map(dic)\n    return data  \n\ndef cabin_last(data):\n    list_cabin_last=data['cabin_last'].to_list()\n    cabin_last_set=set(list_cabin_last)\n    nu_list=[]\n    cabin_last_list=[]\n    \n    for i in range(len(cabin_last_set)):\n        nu_list.append(i)\n        cabin_last_list.append(list(cabin_last_set)[i])\n    dic={}\n    for i in range(len(cabin_last_list)):\n        dic[cabin_last_list[i]]=nu_list[i]\n    \n    data['cabin_last']=data['cabin_last'].map(dic)\n    return data    \n\ndef trans(train):\n    train=vip(train)\n    train=HP(train)\n    train=cabin(train)\n    train=CryoSleep(train)\n    train=Transported(train)\n    train=cabin_first(train)\n    train=cabin_last(train)\n    train=Dest(train)\n    \n    return train\n\ndef Dest(data):\n    list_Dest=data['Destination'].to_list()\n    Dest_set=set(list_Dest)\n    nu_list=[]\n    Dest_list=[]\n\n    for i in range(len(Dest_set)):\n        nu_list.append(i)\n        Dest_list.append(list(Dest_set)[i])\n    dic={}\n    for i in range(len(Dest_list)):\n        dic[Dest_list[i]]=nu_list[i]\n        \n    data['Destination']=data['Destination'].map(dic)\n    return data\n\ntest=pd.read_csv('../input/spaceship-titanic/test.csv') #testデータ読み込み\ntrain=pd.read_csv('../input/spaceship-titanic/train.csv')#trainデータ読み込み\nsub=pd.read_csv('../input/spaceship-titanic/sample_submission.csv')#submissionデータ読み込み\n\ntrain=trans(train)\ntrain\ntest['Transported']=sub['Transported']\n\ntest=trans(test)#テストデータを数字に変換\ntest\n\nimport lightgbm as lgb #lightGBMのライブラリの追加\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import  accuracy_score\n\ntrain_explain=train[['HomePlanet','CryoSleep','Destination','Age',\n                    'cabin_first','cabin_number','cabin_last']].values\n\ntrain_target=train['Transported'].values\n\ntest_explain=test[['HomePlanet','CryoSleep','Destination','Age',\n                   'cabin_first','cabin_number','cabin_last']].values\n\ntest_target=test['Transported'].values\n\nlgb_train=lgb.Dataset(train_explain,train_target)\nlgb_test=lgb.Dataset(test_explain,test_target)\n\n#params={\n#    'task':'train',\n#    'boosting_type':'gbdt',\n#    'objective':'multiclass',\n#    'num_class':2,\n#    'metric':'multi_logloss'\n#}\nparams = {\n'task': 'train',\n'boosting_type': 'gbdt',\n'objective': 'binary', # 目的 : 2クラス分類\n'metric': {'binary_error'}, # 評価指標 : 誤り率(= 1-正答率)\n}\nmodel=lgb.train(params,lgb_train)\n\n#test_target=pd.Series(test_target)\n#test_predicted=model.predict(test_explain,num_iteration=model.best_iteration)\n\n# テストデータの予測 (クラス1の予測確率(クラス1である確率)を返す)\ny_pred_prob = model.predict(test_explain)\n# テストデータの予測 (予測クラス(0 or 1)を返す)\ny_pred = np.where(y_pred_prob < 0.5, 0, 1) # 0.5より小さい場合0 ,そうでない場合1を返す\n\n#y_pred= []\n#for x in test_predicted:\n#    y_pred.append(np.argmax(x))\n\nprint(accuracy_score(y_pred,test_target))\n\npre=pd.DataFrame(test['PassengerId'])\n\npre['Transported']=y_pred\n\npre['Transported']=pre['Transported'].map({0:True,1:False})\npre\npre.to_csv('submission.csv',index=False)\n\n\n#u=train['cabin_number'].to_list()\n#for i in range(len(u)):\n    #u[i]=int(u[i])\n#print(max(u))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T17:17:06.789442Z","iopub.execute_input":"2022-07-06T17:17:06.789877Z","iopub.status.idle":"2022-07-06T17:17:07.147866Z","shell.execute_reply.started":"2022-07-06T17:17:06.789843Z","shell.execute_reply":"2022-07-06T17:17:07.146981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n","metadata":{"execution":{"iopub.status.busy":"2022-07-06T16:45:33.423255Z","iopub.execute_input":"2022-07-06T16:45:33.42376Z","iopub.status.idle":"2022-07-06T16:45:33.441672Z","shell.execute_reply.started":"2022-07-06T16:45:33.42372Z","shell.execute_reply":"2022-07-06T16:45:33.439842Z"},"trusted":true},"execution_count":null,"outputs":[]}]}