{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.mlab as mlab\nimport seaborn as sns\nimport matplotlib\nimport pandas as pd\nimport numpy as np\nfrom numpy import *\nplt.style.use('ggplot')\nfrom matplotlib.pyplot import figure\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nmatplotlib.rcParams['figure.figsize'] = (12,8)\npd.options.mode.chained_assignment = None\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T08:13:15.155800Z","iopub.execute_input":"2022-07-05T08:13:15.156354Z","iopub.status.idle":"2022-07-05T08:13:16.584768Z","shell.execute_reply.started":"2022-07-05T08:13:15.156223Z","shell.execute_reply":"2022-07-05T08:13:16.583854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_data.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:16.586773Z","iopub.execute_input":"2022-07-05T08:13:16.587786Z","iopub.status.idle":"2022-07-05T08:13:16.627543Z","shell.execute_reply.started":"2022-07-05T08:13:16.587750Z","shell.execute_reply":"2022-07-05T08:13:16.626469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Тут мы делаем карту отсутствующих данных\ndef get_heatmap(dataset):\n  cols = dataset.columns\n  colours = ['black', 'red']\n  sns.heatmap(dataset[cols].isnull(), cmap=sns.color_palette(colours))\nget_heatmap(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:16.629211Z","iopub.execute_input":"2022-07-05T08:13:16.629905Z","iopub.status.idle":"2022-07-05T08:13:17.314294Z","shell.execute_reply.started":"2022-07-05T08:13:16.629869Z","shell.execute_reply":"2022-07-05T08:13:17.312433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Удаляем лишние колонки\ndel test_data['Cabin']\ndel test_data['Fare']\ndel test_data['Name']\ntest_data","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.318839Z","iopub.execute_input":"2022-07-05T08:13:17.319944Z","iopub.status.idle":"2022-07-05T08:13:17.350123Z","shell.execute_reply.started":"2022-07-05T08:13:17.319905Z","shell.execute_reply":"2022-07-05T08:13:17.348779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Перевожу Пол в цифровое значение\ntest_data.Sex.unique()\n\nSex =['male', 'female']\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\ndata_new = labelencoder.fit_transform(Sex)\ndata_new","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.351772Z","iopub.execute_input":"2022-07-05T08:13:17.352165Z","iopub.status.idle":"2022-07-05T08:13:17.550508Z","shell.execute_reply.started":"2022-07-05T08:13:17.352129Z","shell.execute_reply":"2022-07-05T08:13:17.549484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Тут мы делаем отбор числовых колонок\ntest_data_numeric = test_data.select_dtypes(include=[np.number])\nnumeric_cols = test_data_numeric.columns.values\nprint(numeric_cols)\n\n#Заполняем пропуски средним возрастным значением\nmed = test_data['Age'].median()\nprint(med)\ntest_data['Age'] = test_data['Age'].fillna(med)\n\nfor col in numeric_cols:\n    missing = test_data[col].isnull()\n    num_missing = np.sum(missing)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.553131Z","iopub.execute_input":"2022-07-05T08:13:17.553663Z","iopub.status.idle":"2022-07-05T08:13:17.578152Z","shell.execute_reply.started":"2022-07-05T08:13:17.553611Z","shell.execute_reply":"2022-07-05T08:13:17.576825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Находим и заполняем буквенные пропуски\ntest_data_non_numeric = test_data.select_dtypes(exclude=[np.number])\nnon_numeric_cols = test_data_non_numeric.columns.values\n\nfor col in non_numeric_cols:\n    missing = test_data[col].isnull()\n    num_missing = np.sum(missing)\n    \n    if num_missing > 0:  \n        print('imputing missing values for: {}'.format(col))\n        test_data['{}_ismissing'.format(col)] = missing\n        \n        top = test_data[col].describe()['top']\n        test_data[col] = test_data[col].fillna(top)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.582156Z","iopub.execute_input":"2022-07-05T08:13:17.582894Z","iopub.status.idle":"2022-07-05T08:13:17.594370Z","shell.execute_reply.started":"2022-07-05T08:13:17.582855Z","shell.execute_reply":"2022-07-05T08:13:17.593371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Тут мы видим, что отсутствующих данных больше нет\nfor col in test_data.columns:\n    pct_missing = np.mean(test_data[col].isnull())\n    print('{} - {}%'.format(col, round(pct_missing*100)))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.598348Z","iopub.execute_input":"2022-07-05T08:13:17.599101Z","iopub.status.idle":"2022-07-05T08:13:17.610211Z","shell.execute_reply.started":"2022-07-05T08:13:17.599051Z","shell.execute_reply":"2022-07-05T08:13:17.608939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ratio_Sex = test_data['Sex'].value_counts()\nfig = plt.figure(figsize=(5, 5))\naxes = fig.add_axes([0, 0, 1, 1])\naxes.pie(ratio_Sex,labels=ratio_Sex.index,autopct='%.1f%%')\naxes.legend(ratio_Sex,title = 'Sex',loc = 'center left',bbox_to_anchor=(1, 0, 0.5, 1.5))\naxes.set(title='Соотношение полов')\naxes.legend().set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.611803Z","iopub.execute_input":"2022-07-05T08:13:17.612192Z","iopub.status.idle":"2022-07-05T08:13:17.764158Z","shell.execute_reply.started":"2022-07-05T08:13:17.612158Z","shell.execute_reply":"2022-07-05T08:13:17.762716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nax.bar(test_data.Age, test_data.PassengerId)\nax.set(title='Возраст')\nax.legend().set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:17.769796Z","iopub.execute_input":"2022-07-05T08:13:17.772187Z","iopub.status.idle":"2022-07-05T08:13:18.850349Z","shell.execute_reply.started":"2022-07-05T08:13:17.772113Z","shell.execute_reply":"2022-07-05T08:13:18.849059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_data.head()\n\n#Удалил колону Name\ndel train_data['Name']\ntrain_data\n\n#Перевожу Пол в цифровое значение\ntrain_data.Sex.unique()\n\nSex =['male', 'female']\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\ndata_new = labelencoder.fit_transform(Sex)\ndata_new\n\n#Тут мы делаем отбор числовых колонок\ntrain_data_numeric = train_data.select_dtypes(include=[np.number])\nnumeric_cols = train_data_numeric.columns.values\nprint(numeric_cols)\n\ntrain_data_non_numeric = train_data.select_dtypes(exclude=[np.number])\nnon_numeric_cols = train_data_non_numeric.columns.values\n\nfor col in non_numeric_cols:\n    missing = train_data[col].isnull()\n    num_missing = np.sum(missing)\n    \n    if num_missing > 0:  \n        print('imputing missing values for: {}'.format(col))\n        train_data['{}_ismissing'.format(col)] = missing\n        \n        top = train_data[col].describe()['top']\n        train_data[col] = train_data[col].fillna(top)\n\n\n#Заполняем пропуски средним возрастным значением\nmed = train_data['Age'].median()\nprint(med)\ntrain_data['Age'] = train_data['Age'].fillna(med)\n\nfor col in numeric_cols:\n    missing = train_data[col].isnull()\n    num_missing = np.sum(missing)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:18.852364Z","iopub.execute_input":"2022-07-05T08:13:18.853987Z","iopub.status.idle":"2022-07-05T08:13:18.901458Z","shell.execute_reply.started":"2022-07-05T08:13:18.853931Z","shell.execute_reply":"2022-07-05T08:13:18.900255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Тут мы видим, что отсутствующих данных больше нет\nfor col in train_data.columns:\n    pct_missing = np.mean(train_data[col].isnull())\n    print('{} - {}%'.format(col, round(pct_missing*100)))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:18.903310Z","iopub.execute_input":"2022-07-05T08:13:18.904047Z","iopub.status.idle":"2022-07-05T08:13:18.917220Z","shell.execute_reply.started":"2022-07-05T08:13:18.904000Z","shell.execute_reply":"2022-07-05T08:13:18.915832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\ny = train_data[\"Survived\"]\n\nfeatures = [\"Pclass\", \"Sex\", \"SibSp\", \"Parch\",\"Age\"]\nX = pd.get_dummies(train_data[features])\nX_test = pd.get_dummies(test_data[features])\n\nmodel = RandomForestClassifier(n_estimators=100, max_depth=5, random_state=1)\nmodel.fit(X, y)\npredictions = model.predict(X_test)\n\noutput = pd.DataFrame({'PassengerId': test_data.PassengerId, 'Survived': predictions})\noutput.to_csv('submission.csv', index=False)\nprint(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T08:13:18.918599Z","iopub.execute_input":"2022-07-05T08:13:18.919682Z","iopub.status.idle":"2022-07-05T08:13:19.527156Z","shell.execute_reply.started":"2022-07-05T08:13:18.919643Z","shell.execute_reply":"2022-07-05T08:13:19.526287Z"},"trusted":true},"execution_count":null,"outputs":[]}]}