{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T05:10:27.703295Z","iopub.execute_input":"2022-07-12T05:10:27.704223Z","iopub.status.idle":"2022-07-12T05:10:27.716373Z","shell.execute_reply.started":"2022-07-12T05:10:27.704169Z","shell.execute_reply":"2022-07-12T05:10:27.715012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/titanic/train.csv')\ntest = pd.read_csv('../input/titanic/test.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:30.572886Z","iopub.execute_input":"2022-07-12T05:10:30.574149Z","iopub.status.idle":"2022-07-12T05:10:30.611561Z","shell.execute_reply.started":"2022-07-12T05:10:30.574094Z","shell.execute_reply":"2022-07-12T05:10:30.610595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:33.215304Z","iopub.execute_input":"2022-07-12T05:10:33.215759Z","iopub.status.idle":"2022-07-12T05:10:33.238379Z","shell.execute_reply.started":"2022-07-12T05:10:33.215726Z","shell.execute_reply":"2022-07-12T05:10:33.237095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def kesson_table(df): \n    null_val = df.isnull().sum()\n    percent = 100 * df.isnull().sum()/len(df)\n    kesson_table = pd.concat([null_val, percent], axis=1)\n    kesson_table_ren_columns = kesson_table.rename(\n    columns = {0 : '欠損数', 1 : '%'})\n    return kesson_table_ren_columns\n\ndisplay(kesson_table(train))\ndisplay(kesson_table(test))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:36.559452Z","iopub.execute_input":"2022-07-12T05:10:36.559843Z","iopub.status.idle":"2022-07-12T05:10:36.592151Z","shell.execute_reply.started":"2022-07-12T05:10:36.559811Z","shell.execute_reply":"2022-07-12T05:10:36.590972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"Age\"] = train[\"Age\"].fillna(train[\"Age\"].median())\ntrain[\"Embarked\"] = train[\"Embarked\"].fillna(\"S\")\nkesson_table(train)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:41.607135Z","iopub.execute_input":"2022-07-12T05:10:41.607597Z","iopub.status.idle":"2022-07-12T05:10:41.635920Z","shell.execute_reply.started":"2022-07-12T05:10:41.607563Z","shell.execute_reply":"2022-07-12T05:10:41.634720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def g_d(df):\n    df = pd.get_dummies(df, columns=['Pclass', 'Sex', 'Embarked'], drop_first=True)\n    return df\n\ntrain_ = g_d(train)\ntrain_.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:44.755007Z","iopub.execute_input":"2022-07-12T05:10:44.755414Z","iopub.status.idle":"2022-07-12T05:10:44.792103Z","shell.execute_reply.started":"2022-07-12T05:10:44.755382Z","shell.execute_reply":"2022-07-12T05:10:44.791213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"Age\"] = test[\"Age\"].fillna(test[\"Age\"].median())\ntest[\"Fare\"] = test[\"Fare\"].fillna(test[\"Fare\"].median())\ndisplay(kesson_table(test))\n\ntest_ = g_d(test)\ntest_.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:47.400960Z","iopub.execute_input":"2022-07-12T05:10:47.401559Z","iopub.status.idle":"2022-07-12T05:10:47.454537Z","shell.execute_reply.started":"2022-07-12T05:10:47.401508Z","shell.execute_reply":"2022-07-12T05:10:47.453480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"select_params = [\n    'Age',\n    'SibSp',\n    'Parch',\n    'Fare',\n    'Pclass_2',\n    'Pclass_3',\n    'Sex_male',\n    'Embarked_Q',\n    'Embarked_S'\n]\n\ntarget = train_['Survived'].values\nfeatures = train_[select_params].values\ntest_features = test_[select_params].values","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:50.719443Z","iopub.execute_input":"2022-07-12T05:10:50.720434Z","iopub.status.idle":"2022-07-12T05:10:50.730624Z","shell.execute_reply.started":"2022-07-12T05:10:50.720393Z","shell.execute_reply":"2022-07-12T05:10:50.729405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import GridSearchCV\n\nLR = LogisticRegression(max_iter=10000)\n\nparams = {'C':[0.01, 0.05, 0.1]}\nCV = GridSearchCV(LR, params)\nCV.fit(features, target)\nprint(CV.best_estimator_)\nprint(\"CV:\",CV.best_score_)\n\nfcst = CV.predict(test_features)\nprint(fcst)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:55.313644Z","iopub.execute_input":"2022-07-12T05:10:55.314130Z","iopub.status.idle":"2022-07-12T05:10:56.330361Z","shell.execute_reply.started":"2022-07-12T05:10:55.314089Z","shell.execute_reply":"2022-07-12T05:10:56.329097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nRF = RandomForestClassifier(n_jobs=-1)\n\nparams = {'max_depth':[5, 10, 50], 'n_estimators': [200, 250, 300]}\nCV = GridSearchCV(RF, params)\nCV.fit(features, target)\nprint(CV.best_estimator_)\nprint(\"CV:\",CV.best_score_)\n\nfcst2 = CV.predict(test_features)\nprint(fcst2)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:10:59.383907Z","iopub.execute_input":"2022-07-12T05:10:59.384396Z","iopub.status.idle":"2022-07-12T05:11:26.259249Z","shell.execute_reply.started":"2022-07-12T05:10:59.384360Z","shell.execute_reply":"2022-07-12T05:11:26.258005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# PassengerIdを取得\nPassengerId = np.array(test[\"PassengerId\"]).astype(int)\n# my_prediction(予測データ）とPassengerIdをデータフレームへ落とし込む\nsolution = pd.DataFrame(fcst2, PassengerId, columns = [\"Survived\"])\n# titanic.csvとして書き出し\nsolution.to_csv(\"titanic.csv\", index_label = [\"PassengerId\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T05:23:59.480884Z","iopub.execute_input":"2022-07-12T05:23:59.481374Z","iopub.status.idle":"2022-07-12T05:23:59.491757Z","shell.execute_reply.started":"2022-07-12T05:23:59.481338Z","shell.execute_reply":"2022-07-12T05:23:59.490851Z"},"trusted":true},"execution_count":null,"outputs":[]}]}