{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\nimport missingno as msno\nfrom datetime import date\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neighbors import LocalOutlierFactor\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder, StandardScaler, RobustScaler\n\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)\npd.set_option('display.float_format', lambda x: '%.3f' % x)\npd.set_option('display.width', 500)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T18:54:07.500543Z","iopub.execute_input":"2022-08-01T18:54:07.500909Z","iopub.status.idle":"2022-08-01T18:54:07.507678Z","shell.execute_reply.started":"2022-08-01T18:54:07.500879Z","shell.execute_reply":"2022-08-01T18:54:07.506720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.577841Z","iopub.execute_input":"2022-08-01T18:54:07.578885Z","iopub.status.idle":"2022-08-01T18:54:07.599443Z","shell.execute_reply.started":"2022-08-01T18:54:07.578847Z","shell.execute_reply":"2022-08-01T18:54:07.598544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.661602Z","iopub.execute_input":"2022-08-01T18:54:07.662802Z","iopub.status.idle":"2022-08-01T18:54:07.681445Z","shell.execute_reply.started":"2022-08-01T18:54:07.662767Z","shell.execute_reply":"2022-08-01T18:54:07.679490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gender_submission = pd.read_csv(\"/kaggle/input/titanic/gender_submission.csv\")\ngender_submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.740658Z","iopub.execute_input":"2022-08-01T18:54:07.741023Z","iopub.status.idle":"2022-08-01T18:54:07.752992Z","shell.execute_reply.started":"2022-08-01T18:54:07.740991Z","shell.execute_reply":"2022-08-01T18:54:07.751897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe().T","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.809340Z","iopub.execute_input":"2022-08-01T18:54:07.809711Z","iopub.status.idle":"2022-08-01T18:54:07.839396Z","shell.execute_reply.started":"2022-08-01T18:54:07.809679Z","shell.execute_reply":"2022-08-01T18:54:07.838317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Numerical and Categorical değişkenlerin yakalanması**","metadata":{}},{"cell_type":"code","source":"def grab_col_names(dataframe, cat_th=10, car_th=20):\n    \"\"\"\n    Veri setindeki kategorik, numerik ve kategorik fakat kardinal değişkenlerin isimlerini verir.\n    Not: Kategorik değişkenlerin içerisine numerik görünümlü kategorik değişkenler de dahildir.\n    Parameters\n    ------\n        dataframe: dataframe\n                Değişken isimleri alınmak istenilen dataframe\n        cat_th: int, optional\n                numerik fakat kategorik olan değişkenler için sınıf eşik değeri\n        car_th: int, optional\n                kategorik fakat kardinal değişkenler için sınıf eşik değeri\n    Returns\n    ------\n        cat_cols: list\n                Kategorik değişken listesi\n        num_cols: list\n                Numerik değişken listesi\n        cat_but_car: list\n                Kategorik görünümlü kardinal değişken listesi\n    Examples\n    ------\n        import seaborn as sns\n        df = sns.load_dataset(\"iris\")\n        print(grab_col_names(df))\n    Notes\n    ------\n        cat_cols + num_cols + cat_but_car = toplam değişken sayısı\n        num_but_cat cat_cols'un içerisinde.\n    \"\"\"\n    # cat_cols, cat_but_car\n    cat_cols = [col for col in dataframe.columns if dataframe[col].dtypes == \"O\"]\n    num_but_cat = [col for col in dataframe.columns if\n                   dataframe[col].nunique() < cat_th and dataframe[col].dtypes != \"O\"]\n    cat_but_car = [col for col in dataframe.columns if\n                   dataframe[col].nunique() > car_th and dataframe[col].dtypes == \"O\"]\n    cat_cols = cat_cols + num_but_cat\n    cat_cols = [col for col in cat_cols if col not in cat_but_car]\n\n    # num_cols\n    num_cols = [col for col in dataframe.columns if dataframe[col].dtypes != \"O\"]\n    num_cols = [col for col in num_cols if col not in num_but_cat]\n\n    print(f\"Observations: {dataframe.shape[0]}\")\n    print(f\"Variables: {dataframe.shape[1]}\")\n    print(f'cat_cols: {len(cat_cols)}')\n    print(f'num_cols: {len(num_cols)}')\n    print(f'cat_but_car: {len(cat_but_car)}')\n    print(f'num_but_cat: {len(num_but_cat)}')\n\n    return cat_cols, num_cols, cat_but_car\n\n\ncat_cols, num_cols, cat_but_car = grab_col_names(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.870514Z","iopub.execute_input":"2022-08-01T18:54:07.870843Z","iopub.status.idle":"2022-08-01T18:54:07.884162Z","shell.execute_reply.started":"2022-08-01T18:54:07.870815Z","shell.execute_reply":"2022-08-01T18:54:07.883330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categorical Veriables Analysis","metadata":{}},{"cell_type":"code","source":"def cat_summary(dataframe, col_name, plot=False):\n    print(pd.DataFrame({col_name: dataframe[col_name].value_counts(),\n                        \"Ratio\": 100 * dataframe[col_name].value_counts() / len(dataframe)}))\n    print(\"##########################################\")\n    if plot:\n        sns.countplot(x=dataframe[col_name], data=dataframe)\n        plt.show(block=True)\n\n\nfor col in cat_cols:\n    cat_summary(train_data, col, plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:07.906033Z","iopub.execute_input":"2022-08-01T18:54:07.906839Z","iopub.status.idle":"2022-08-01T18:54:09.106100Z","shell.execute_reply.started":"2022-08-01T18:54:07.906801Z","shell.execute_reply":"2022-08-01T18:54:09.105084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Numerical Veriables Analysis**","metadata":{}},{"cell_type":"code","source":"def num_summary(dataframe, numerical_col, plot=False):\n    quantiles = [0.05, 0.10, 0.20, 0.30, 0.40, 0.50, 0.60, 0.70, 0.80, 0.90, 0.95, 0.99]\n    print(dataframe[numerical_col].describe(quantiles).T)\n\n    if plot:\n        dataframe[numerical_col].hist(bins=20)\n        plt.xlabel(numerical_col)\n        plt.title(numerical_col)\n        plt.show(block=True)\n\n\nfor col in num_cols:\n    num_summary(train_data, col, plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:09.108604Z","iopub.execute_input":"2022-08-01T18:54:09.109346Z","iopub.status.idle":"2022-08-01T18:54:09.889188Z","shell.execute_reply.started":"2022-08-01T18:54:09.109313Z","shell.execute_reply":"2022-08-01T18:54:09.888171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Num veriables analysis for Target","metadata":{}},{"cell_type":"code","source":"def target_summary_with_num(dataframe, target, numerical_col):\n    print(dataframe.groupby(target).agg({numerical_col: \"mean\"}), end=\"\\n\\n\\n\")\n\n\nfor col in num_cols:\n    target_summary_with_num(train_data, \"Survived\", col)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:09.890571Z","iopub.execute_input":"2022-08-01T18:54:09.890894Z","iopub.status.idle":"2022-08-01T18:54:09.905132Z","shell.execute_reply.started":"2022-08-01T18:54:09.890865Z","shell.execute_reply":"2022-08-01T18:54:09.903966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Passanger Id is a looks useless**","metadata":{}},{"cell_type":"markdown","source":"Categorical Veriables analysis for target","metadata":{}},{"cell_type":"code","source":"def target_summary_with_cat(dataframe, target, categorical_col):\n    print(pd.DataFrame({\"TARGET_MEAN\": dataframe.groupby(categorical_col)[target].mean(),\n                        \"TARGET_COUNT\": dataframe.groupby(categorical_col)[target].count()}), end=\"\\n\\n\\n\")\n\n\nfor col in cat_cols:\n    target_summary_with_cat(train_data, \"Survived\", col)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:09.907169Z","iopub.execute_input":"2022-08-01T18:54:09.907447Z","iopub.status.idle":"2022-08-01T18:54:09.930866Z","shell.execute_reply.started":"2022-08-01T18:54:09.907415Z","shell.execute_reply":"2022-08-01T18:54:09.929997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Correlation**","metadata":{}},{"cell_type":"code","source":"train_data.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:09.931866Z","iopub.execute_input":"2022-08-01T18:54:09.932648Z","iopub.status.idle":"2022-08-01T18:54:09.943658Z","shell.execute_reply.started":"2022-08-01T18:54:09.932610Z","shell.execute_reply":"2022-08-01T18:54:09.942860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def high_correlated_cols(dataframe, plot=False, corr_th=0.95):\n    corr = dataframe.corr()\n    corr_matrix = corr.abs()\n    upper_triangle_matrix = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(np.bool))\n    drop_list = [col for col in upper_triangle_matrix.columns if any(upper_triangle_matrix[col] > 0.90)]\n    if plot:\n        import seaborn as sns\n        import matplotlib.pyplot as plt\n        sns.set(rc={\"figure.figsize\": (15, 15)})\n        sns.heatmap(corr, cmap=\"RdBu\")\n        plt.show()\n    return drop_list\n\n\nhigh_correlated_cols(train_data, plot=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:09.944882Z","iopub.execute_input":"2022-08-01T18:54:09.945289Z","iopub.status.idle":"2022-08-01T18:54:10.251538Z","shell.execute_reply.started":"2022-08-01T18:54:09.945262Z","shell.execute_reply":"2022-08-01T18:54:10.250574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_list = high_correlated_cols(train_data)\n\nf, ax = plt.subplots(figsize=[18, 13])\nsns.heatmap(train_data[num_cols].corr(), annot=True, fmt=\".2f\", ax=ax, )\nax.set_title(\"Correlation Matrix\", fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.253033Z","iopub.execute_input":"2022-08-01T18:54:10.253656Z","iopub.status.idle":"2022-08-01T18:54:10.489267Z","shell.execute_reply.started":"2022-08-01T18:54:10.253614Z","shell.execute_reply":"2022-08-01T18:54:10.488266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Missing Value Analysis**","metadata":{}},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.490318Z","iopub.execute_input":"2022-08-01T18:54:10.490603Z","iopub.status.idle":"2022-08-01T18:54:10.499061Z","shell.execute_reply.started":"2022-08-01T18:54:10.490563Z","shell.execute_reply":"2022-08-01T18:54:10.498174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def missing_values_table(dataframe, na_name=False):\n    na_columns = [col for col in dataframe.columns if dataframe[col].isnull().sum() > 0]\n    n_miss = dataframe[na_columns].isnull().sum().sort_values(ascending=False)\n    ratio = (dataframe[na_columns].isnull().sum() / dataframe.shape[0] * 100).sort_values(ascending=False)\n    missing_df = pd.concat([n_miss, np.round(ratio, 2)], axis=1, keys=['n_miss', 'ratio'])\n    print(missing_df, end=\"\\n\")\n    if na_name:\n        return na_columns\n\n\nna_columns = missing_values_table(train_data, na_name=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.500209Z","iopub.execute_input":"2022-08-01T18:54:10.500508Z","iopub.status.idle":"2022-08-01T18:54:10.521231Z","shell.execute_reply.started":"2022-08-01T18:54:10.500481Z","shell.execute_reply":"2022-08-01T18:54:10.520297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(\"Cabin\",inplace = True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.524291Z","iopub.execute_input":"2022-08-01T18:54:10.524571Z","iopub.status.idle":"2022-08-01T18:54:10.529247Z","shell.execute_reply.started":"2022-08-01T18:54:10.524545Z","shell.execute_reply":"2022-08-01T18:54:10.528290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"Age\"] = train_data[\"Age\"].fillna(train_data.groupby(\"Sex\")[\"Age\"].transform(\"median\"))\nmissing_values_table(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.530574Z","iopub.execute_input":"2022-08-01T18:54:10.530875Z","iopub.status.idle":"2022-08-01T18:54:10.549720Z","shell.execute_reply.started":"2022-08-01T18:54:10.530849Z","shell.execute_reply":"2022-08-01T18:54:10.548763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.apply(lambda x: x.fillna(x.mode()[0]) if (x.dtype == \"O\" and len(x.unique()) <= 10) else x, axis=0)\nmissing_values_table(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.550974Z","iopub.execute_input":"2022-08-01T18:54:10.551237Z","iopub.status.idle":"2022-08-01T18:54:10.565644Z","shell.execute_reply.started":"2022-08-01T18:54:10.551212Z","shell.execute_reply":"2022-08-01T18:54:10.564997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.566608Z","iopub.execute_input":"2022-08-01T18:54:10.566880Z","iopub.status.idle":"2022-08-01T18:54:10.579454Z","shell.execute_reply.started":"2022-08-01T18:54:10.566854Z","shell.execute_reply":"2022-08-01T18:54:10.578506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One Hot Encoding","metadata":{}},{"cell_type":"code","source":"def one_hot_encoder(dataframe, categorical_cols, drop_first = False):\n    dataframe = pd.get_dummies(dataframe, columns=categorical_cols,drop_first=drop_first)\n    return dataframe","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.580523Z","iopub.execute_input":"2022-08-01T18:54:10.580801Z","iopub.status.idle":"2022-08-01T18:54:10.587740Z","shell.execute_reply.started":"2022-08-01T18:54:10.580776Z","shell.execute_reply":"2022-08-01T18:54:10.587016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ohe_cols = [col for col in train_data.columns if 10 >= train_data[col].nunique() > 2]\n\ntrain_data = one_hot_encoder(train_data, ohe_cols)\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.588886Z","iopub.execute_input":"2022-08-01T18:54:10.589126Z","iopub.status.idle":"2022-08-01T18:54:10.616225Z","shell.execute_reply.started":"2022-08-01T18:54:10.589103Z","shell.execute_reply":"2022-08-01T18:54:10.615568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Label Encoding","metadata":{}},{"cell_type":"code","source":"def label_encoder(dataframe, binary_col):\n    labelencoder = LabelEncoder()\n    dataframe[binary_col] = labelencoder.fit_transform((dataframe[binary_col]))\n    return  dataframe","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.617069Z","iopub.execute_input":"2022-08-01T18:54:10.617578Z","iopub.status.idle":"2022-08-01T18:54:10.622309Z","shell.execute_reply.started":"2022-08-01T18:54:10.617550Z","shell.execute_reply":"2022-08-01T18:54:10.621315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"binary_cols = [col for col in train_data.columns if train_data[col].dtype not in[int, float]\n                and train_data[col].nunique() == 2]\n\nbinary_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.623919Z","iopub.execute_input":"2022-08-01T18:54:10.624659Z","iopub.status.idle":"2022-08-01T18:54:10.637872Z","shell.execute_reply.started":"2022-08-01T18:54:10.624620Z","shell.execute_reply":"2022-08-01T18:54:10.637164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in binary_cols:\n    train_data = label_encoder(train_data,col)\n\nbinary_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.639470Z","iopub.execute_input":"2022-08-01T18:54:10.640313Z","iopub.status.idle":"2022-08-01T18:54:10.654535Z","shell.execute_reply.started":"2022-08-01T18:54:10.640256Z","shell.execute_reply":"2022-08-01T18:54:10.653808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape\n\ncat_cols, num_cols, cat_but_car = grab_col_names(train_data)\n\nnum_cols = [col for col in num_cols if \"PassengerId\" not in col]\n\nnum_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.656061Z","iopub.execute_input":"2022-08-01T18:54:10.656339Z","iopub.status.idle":"2022-08-01T18:54:10.670208Z","shell.execute_reply.started":"2022-08-01T18:54:10.656315Z","shell.execute_reply":"2022-08-01T18:54:10.669402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Rare Analysis**","metadata":{}},{"cell_type":"code","source":"def rare_analyser(dataframe, target, cat_cols):\n    for col in cat_cols:\n        print(col,\":\",len(dataframe[col].value_counts()))\n        print(pd.DataFrame({\"COUNT\": dataframe[col].value_counts(),\n                            \"RATIO\": dataframe[col].value_counts() / len(dataframe),\n                            \"TARGET_MEAN\":dataframe.groupby(col)[target].mean()}),end=\"\\n\\n\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.671197Z","iopub.execute_input":"2022-08-01T18:54:10.671464Z","iopub.status.idle":"2022-08-01T18:54:10.677456Z","shell.execute_reply.started":"2022-08-01T18:54:10.671431Z","shell.execute_reply":"2022-08-01T18:54:10.676401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rare_analyser(train_data, \"Survived\",cat_cols)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.679509Z","iopub.execute_input":"2022-08-01T18:54:10.680336Z","iopub.status.idle":"2022-08-01T18:54:10.763623Z","shell.execute_reply.started":"2022-08-01T18:54:10.680293Z","shell.execute_reply":"2022-08-01T18:54:10.762621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"useless_cols = [col for col in train_data.columns if train_data[col].nunique()  == 2 and\n                (train_data[col].value_counts() / len(train_data) < 0.01).any(axis=None)]\n\nuseless_cols","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.764719Z","iopub.execute_input":"2022-08-01T18:54:10.765074Z","iopub.status.idle":"2022-08-01T18:54:10.786751Z","shell.execute_reply.started":"2022-08-01T18:54:10.765047Z","shell.execute_reply":"2022-08-01T18:54:10.785958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(useless_cols, axis=1,inplace=True)\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T18:54:10.788171Z","iopub.execute_input":"2022-08-01T18:54:10.788789Z","iopub.status.idle":"2022-08-01T18:54:10.806049Z","shell.execute_reply.started":"2022-08-01T18:54:10.788747Z","shell.execute_reply":"2022-08-01T18:54:10.805138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"y = train_data[\"Survived\"]\nX = train_data.drop([\"PassengerId\",\"Survived\",\"Name\",\"Ticket\"],axis=1)\n\nX_train ,X_test, y_train , y_test = train_test_split(X,y,test_size=0.30, random_state=17)\n\nfrom sklearn.ensemble import RandomForestClassifier\n\nrf_model = RandomForestClassifier(random_state=46).fit(X_train, y_train)\ny_pred = rf_model.predict(X_test)\naccuracy_score(y_pred, y_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:01:04.129877Z","iopub.execute_input":"2022-08-01T19:01:04.130705Z","iopub.status.idle":"2022-08-01T19:01:04.314087Z","shell.execute_reply.started":"2022-08-01T19:01:04.130665Z","shell.execute_reply":"2022-08-01T19:01:04.313003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**İf no action was taken score**","metadata":{}},{"cell_type":"code","source":"dff = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ndff.dropna(inplace=True)\ndff = pd.get_dummies(dff, columns=[\"Sex\", \"Embarked\"], drop_first=True)\ny = dff[\"Survived\"]\nX = dff.drop([\"PassengerId\", \"Survived\", \"Name\", \"Ticket\", \"Cabin\"], axis=1)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.30, random_state=17)\nrf_model = RandomForestClassifier(random_state=46).fit(X_train, y_train)\ny_pred = rf_model.predict(X_test)\naccuracy_score(y_pred, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T19:03:06.339045Z","iopub.execute_input":"2022-08-01T19:03:06.339499Z","iopub.status.idle":"2022-08-01T19:03:06.531750Z","shell.execute_reply.started":"2022-08-01T19:03:06.339450Z","shell.execute_reply":"2022-08-01T19:03:06.530760Z"},"trusted":true},"execution_count":null,"outputs":[]}]}