{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T05:40:30.947655Z","iopub.execute_input":"2022-07-20T05:40:30.948217Z","iopub.status.idle":"2022-07-20T05:40:30.978754Z","shell.execute_reply.started":"2022-07-20T05:40:30.948108Z","shell.execute_reply":"2022-07-20T05:40:30.977864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = '/kaggle/input/spaceship-titanic/'\ndf = pd.read_csv(os.path.join(path, 'train.csv'))\nsub = pd.read_csv(os.path.join(path, 'sample_submission.csv'))\ntest = pd.read_csv(os.path.join(path, 'test.csv'))\n\ndef cut_unnecesay(df):\n    name = df.Name.fillna('None None').apply(lambda x: x.split()[0])\n    surname = df.Name.fillna('None None').apply(lambda x: x.split()[1])\n    sur, count = np.unique(surname, return_counts=True)\n    dsur = {sur[i]: count[i] for i in range(sur.shape[0])}\n    df[['hEarth', 'hEuropa', 'hMars']] = pd.get_dummies(df.HomePlanet)\n    df[['dCancri', 'dPSO', 'dTRAPPIST']] = pd.get_dummies(df.Destination)\n    df.Age = df.Age.fillna(df.Age.median())\n    df.RoomService = df.RoomService.fillna(0.0)\n    df.ShoppingMall = df.ShoppingMall.fillna(0.0)\n    df.FoodCourt = df.FoodCourt.fillna(0.0)\n    df.Spa = df.Spa.fillna(0.0)\n    df.VRDeck = df.VRDeck.fillna(0.0)\n    df.CryoSleep = df.CryoSleep.replace({\n        False : 0,\n        True : 1,\n        np.nan : 0\n    })\n    df['Family'] = (np.array(list(map(dsur.get, surname))) > 1).astype(int)\n    df.VIP = df.VIP.replace({\n        False : 0,\n        True : 1,\n        np.nan : 0\n    })\n    deck = df.Cabin.fillna('//').apply(lambda x: x.split('/')[0])\n    num = df.Cabin.fillna('//').apply(lambda x: x.split('/')[1])\n    side = df.Cabin.fillna('//').apply(lambda x: x.split('/')[2])\n    df['cNum'] = num\n    df['cDeck'] = deck\n    df['cSide'] = side\n    df['cDeck'] = df['cDeck'].replace({\n        '': 5,\n        'A': 6,\n        'B': 4,\n        'C': 8,\n        'D': 1,\n        'E': 0,\n        'F': 7,\n        'G': 3,\n        'T': 2\n    })\n\n\n    df['cSide'] = df['cSide'].replace({\n        'P': 0,\n        '': 1,\n        'S': 1\n    })\n    df.cNum = df.cNum.replace({'':np.nan}).astype(float)\n    df.cNum = df.cNum.fillna(df.cNum.dropna().median())\n    df = df.drop(['HomePlanet', 'Destination', 'Cabin', 'Name', 'PassengerId'], axis=1)\n    \n    return df\n\ndf = cut_unnecesay(df)\ntest = cut_unnecesay(test)\nX = df.drop(['Transported'], axis=1)\ny = df['Transported'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:39:47.704348Z","iopub.execute_input":"2022-07-20T07:39:47.704773Z","iopub.status.idle":"2022-07-20T07:39:47.934631Z","shell.execute_reply.started":"2022-07-20T07:39:47.704737Z","shell.execute_reply":"2022-07-20T07:39:47.933336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.33, random_state=42)\nmodel = XGBClassifier(n_estimators=300, max_depth=3, learning_rate=0.1)\nmodel.fit(X_train, y_train)\nprint(accuracy_score(model.predict(X_test), y_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:39:49.745803Z","iopub.execute_input":"2022-07-20T07:39:49.746226Z","iopub.status.idle":"2022-07-20T07:39:51.909590Z","shell.execute_reply.started":"2022-07-20T07:39:49.746192Z","shell.execute_reply":"2022-07-20T07:39:51.908567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['Transported'] = np.array(model.predict(test)).astype(bool)\nsub.to_csv('sub.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:22:30.130348Z","iopub.execute_input":"2022-07-20T07:22:30.131324Z","iopub.status.idle":"2022-07-20T07:22:30.167916Z","shell.execute_reply.started":"2022-07-20T07:22:30.131279Z","shell.execute_reply":"2022-07-20T07:22:30.166970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Family.corr(df.Transported)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:40:06.949184Z","iopub.execute_input":"2022-07-20T07:40:06.949740Z","iopub.status.idle":"2022-07-20T07:40:06.960744Z","shell.execute_reply.started":"2022-07-20T07:40:06.949691Z","shell.execute_reply":"2022-07-20T07:40:06.959246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"name = df.Name.fillna('None None').apply(lambda x: x.split()[0])\nsurname = df.Name.fillna('None None').apply(lambda x: x.split()[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:28:02.455056Z","iopub.execute_input":"2022-07-20T07:28:02.455491Z","iopub.status.idle":"2022-07-20T07:28:02.475774Z","shell.execute_reply.started":"2022-07-20T07:28:02.455457Z","shell.execute_reply":"2022-07-20T07:28:02.474443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"sur, count = np.unique(surname, return_counts=True)\ndsur = {sur[i]: count[i] for i in range(sur.shape[0])}\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:35:14.811194Z","iopub.execute_input":"2022-07-20T07:35:14.811766Z","iopub.status.idle":"2022-07-20T07:35:14.831769Z","shell.execute_reply.started":"2022-07-20T07:35:14.811722Z","shell.execute_reply":"2022-07-20T07:35:14.830137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:36:55.833916Z","iopub.execute_input":"2022-07-20T07:36:55.834324Z","iopub.status.idle":"2022-07-20T07:36:55.845506Z","shell.execute_reply.started":"2022-07-20T07:36:55.834291Z","shell.execute_reply":"2022-07-20T07:36:55.844491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(df.CryoSleep.replace({\n    False : 0,\n    True : 1,\n    np.nan : -1\n}).to_numpy(), return_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T06:59:18.428795Z","iopub.execute_input":"2022-07-20T06:59:18.429531Z","iopub.status.idle":"2022-07-20T06:59:18.444681Z","shell.execute_reply.started":"2022-07-20T06:59:18.429489Z","shell.execute_reply":"2022-07-20T06:59:18.443570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.cNum.replace({'':np.nan}).astype(float).dropna().median()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T06:09:07.835845Z","iopub.execute_input":"2022-07-20T06:09:07.837219Z","iopub.status.idle":"2022-07-20T06:09:07.854306Z","shell.execute_reply.started":"2022-07-20T06:09:07.837153Z","shell.execute_reply":"2022-07-20T06:09:07.853000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.get_dummies(df.Destination)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T05:54:36.013398Z","iopub.execute_input":"2022-07-20T05:54:36.013935Z","iopub.status.idle":"2022-07-20T05:54:36.034516Z","shell.execute_reply.started":"2022-07-20T05:54:36.013869Z","shell.execute_reply":"2022-07-20T05:54:36.033374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"df.Name","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:24:15.137449Z","iopub.execute_input":"2022-07-20T07:24:15.137861Z","iopub.status.idle":"2022-07-20T07:24:15.160340Z","shell.execute_reply.started":"2022-07-20T07:24:15.137827Z","shell.execute_reply":"2022-07-20T07:24:15.159094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.Age","metadata":{"execution":{"iopub.status.busy":"2022-07-20T07:22:17.444315Z","iopub.execute_input":"2022-07-20T07:22:17.444699Z","iopub.status.idle":"2022-07-20T07:22:17.455665Z","shell.execute_reply.started":"2022-07-20T07:22:17.444669Z","shell.execute_reply":"2022-07-20T07:22:17.454350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}