{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:36.701187Z","iopub.execute_input":"2022-07-11T11:07:36.701640Z","iopub.status.idle":"2022-07-11T11:07:39.006088Z","shell.execute_reply.started":"2022-07-11T11:07:36.701545Z","shell.execute_reply":"2022-07-11T11:07:39.004550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.010678Z","iopub.execute_input":"2022-07-11T11:07:39.011136Z","iopub.status.idle":"2022-07-11T11:07:39.022230Z","shell.execute_reply.started":"2022-07-11T11:07:39.011087Z","shell.execute_reply":"2022-07-11T11:07:39.020783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.024217Z","iopub.execute_input":"2022-07-11T11:07:39.025115Z","iopub.status.idle":"2022-07-11T11:07:39.047606Z","shell.execute_reply.started":"2022-07-11T11:07:39.025069Z","shell.execute_reply":"2022-07-11T11:07:39.046522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.049683Z","iopub.execute_input":"2022-07-11T11:07:39.050548Z","iopub.status.idle":"2022-07-11T11:07:39.083858Z","shell.execute_reply.started":"2022-07-11T11:07:39.050501Z","shell.execute_reply":"2022-07-11T11:07:39.082409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.091805Z","iopub.execute_input":"2022-07-11T11:07:39.092855Z","iopub.status.idle":"2022-07-11T11:07:39.141597Z","shell.execute_reply.started":"2022-07-11T11:07:39.092808Z","shell.execute_reply":"2022-07-11T11:07:39.140472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.143266Z","iopub.execute_input":"2022-07-11T11:07:39.144097Z","iopub.status.idle":"2022-07-11T11:07:39.157000Z","shell.execute_reply.started":"2022-07-11T11:07:39.144053Z","shell.execute_reply":"2022-07-11T11:07:39.155720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.158817Z","iopub.execute_input":"2022-07-11T11:07:39.169662Z","iopub.status.idle":"2022-07-11T11:07:39.217144Z","shell.execute_reply.started":"2022-07-11T11:07:39.169598Z","shell.execute_reply":"2022-07-11T11:07:39.215974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.218405Z","iopub.execute_input":"2022-07-11T11:07:39.218815Z","iopub.status.idle":"2022-07-11T11:07:39.268922Z","shell.execute_reply.started":"2022-07-11T11:07:39.218766Z","shell.execute_reply":"2022-07-11T11:07:39.267599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x=df.Pclass, y=df.Fare)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.274765Z","iopub.execute_input":"2022-07-11T11:07:39.276167Z","iopub.status.idle":"2022-07-11T11:07:39.569232Z","shell.execute_reply.started":"2022-07-11T11:07:39.276120Z","shell.execute_reply":"2022-07-11T11:07:39.567741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0,len(df['Cabin'])):\n    if not pd.isna(df['Cabin'][i]):\n        df['Cabin'][i] = df['Cabin'][i][0]\n    else:\n        pass","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.570543Z","iopub.execute_input":"2022-07-11T11:07:39.571036Z","iopub.status.idle":"2022-07-11T11:07:39.704307Z","shell.execute_reply.started":"2022-07-11T11:07:39.571007Z","shell.execute_reply":"2022-07-11T11:07:39.702641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.706402Z","iopub.execute_input":"2022-07-11T11:07:39.707153Z","iopub.status.idle":"2022-07-11T11:07:39.734440Z","shell.execute_reply.started":"2022-07-11T11:07:39.707105Z","shell.execute_reply":"2022-07-11T11:07:39.733328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\ndf_1 = df.copy()\n\noriginal = df_1\nmask = df_1['Cabin'].isnull()\n\ndf_1 = df_1.astype(str).apply(LabelEncoder().fit_transform)\ndf['Cabin'] = df_1.where(~mask, original)['Cabin']","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.735801Z","iopub.execute_input":"2022-07-11T11:07:39.736108Z","iopub.status.idle":"2022-07-11T11:07:39.900233Z","shell.execute_reply.started":"2022-07-11T11:07:39.736081Z","shell.execute_reply":"2022-07-11T11:07:39.899288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.902669Z","iopub.execute_input":"2022-07-11T11:07:39.903621Z","iopub.status.idle":"2022-07-11T11:07:39.924934Z","shell.execute_reply.started":"2022-07-11T11:07:39.903573Z","shell.execute_reply":"2022-07-11T11:07:39.923433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ny_train = df_1.where(~mask, original)['Cabin'].dropna().values\nX_train = df[df['Cabin'].isna() == False]['Fare'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:39.932419Z","iopub.execute_input":"2022-07-11T11:07:39.932959Z","iopub.status.idle":"2022-07-11T11:07:40.004458Z","shell.execute_reply.started":"2022-07-11T11:07:39.932899Z","shell.execute_reply":"2022-07-11T11:07:40.003327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.005804Z","iopub.execute_input":"2022-07-11T11:07:40.006213Z","iopub.status.idle":"2022-07-11T11:07:40.014503Z","shell.execute_reply.started":"2022-07-11T11:07:40.006173Z","shell.execute_reply":"2022-07-11T11:07:40.013300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LinearRegression\nregressor = LinearRegression()\nregressor.fit(X_train.reshape(-1, 1), y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.015970Z","iopub.execute_input":"2022-07-11T11:07:40.017189Z","iopub.status.idle":"2022-07-11T11:07:40.115159Z","shell.execute_reply.started":"2022-07-11T11:07:40.017143Z","shell.execute_reply":"2022-07-11T11:07:40.114060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(0, len(df['Cabin'])):\n    if pd.isna(df['Cabin'][i]):\n        df['Cabin'][i] = regressor.predict([[df['Fare'][i]]]).round().astype(int)[0]\n    else:\n        pass","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-11T11:07:40.116897Z","iopub.execute_input":"2022-07-11T11:07:40.117188Z","iopub.status.idle":"2022-07-11T11:07:40.507702Z","shell.execute_reply.started":"2022-07-11T11:07:40.117162Z","shell.execute_reply":"2022-07-11T11:07:40.497435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.509704Z","iopub.execute_input":"2022-07-11T11:07:40.510049Z","iopub.status.idle":"2022-07-11T11:07:40.529641Z","shell.execute_reply.started":"2022-07-11T11:07:40.510020Z","shell.execute_reply":"2022-07-11T11:07:40.528472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(subset=['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.531051Z","iopub.execute_input":"2022-07-11T11:07:40.531411Z","iopub.status.idle":"2022-07-11T11:07:40.538952Z","shell.execute_reply.started":"2022-07-11T11:07:40.531381Z","shell.execute_reply":"2022-07-11T11:07:40.537937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ndf['Sex'] = le.fit_transform(df['Sex'])\ndf['Embarked'] = le.fit_transform(df['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.540370Z","iopub.execute_input":"2022-07-11T11:07:40.540928Z","iopub.status.idle":"2022-07-11T11:07:40.553638Z","shell.execute_reply.started":"2022-07-11T11:07:40.540893Z","shell.execute_reply":"2022-07-11T11:07:40.552767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.554731Z","iopub.execute_input":"2022-07-11T11:07:40.555205Z","iopub.status.idle":"2022-07-11T11:07:40.574775Z","shell.execute_reply.started":"2022-07-11T11:07:40.555174Z","shell.execute_reply":"2022-07-11T11:07:40.573381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.drop(['Name','Ticket'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.576424Z","iopub.execute_input":"2022-07-11T11:07:40.576776Z","iopub.status.idle":"2022-07-11T11:07:40.586292Z","shell.execute_reply.started":"2022-07-11T11:07:40.576738Z","shell.execute_reply":"2022-07-11T11:07:40.585011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Embarked'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.588304Z","iopub.execute_input":"2022-07-11T11:07:40.588730Z","iopub.status.idle":"2022-07-11T11:07:40.601000Z","shell.execute_reply.started":"2022-07-11T11:07:40.588688Z","shell.execute_reply":"2022-07-11T11:07:40.599994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.602809Z","iopub.execute_input":"2022-07-11T11:07:40.603225Z","iopub.status.idle":"2022-07-11T11:07:40.634850Z","shell.execute_reply.started":"2022-07-11T11:07:40.603184Z","shell.execute_reply":"2022-07-11T11:07:40.624359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'] = df['Age'].fillna(value=df['Age'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.636785Z","iopub.execute_input":"2022-07-11T11:07:40.637492Z","iopub.status.idle":"2022-07-11T11:07:40.646009Z","shell.execute_reply.started":"2022-07-11T11:07:40.637443Z","shell.execute_reply":"2022-07-11T11:07:40.644866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Age'] = df['Age'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.647207Z","iopub.execute_input":"2022-07-11T11:07:40.647638Z","iopub.status.idle":"2022-07-11T11:07:40.661064Z","shell.execute_reply.started":"2022-07-11T11:07:40.647598Z","shell.execute_reply":"2022-07-11T11:07:40.659777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.663444Z","iopub.execute_input":"2022-07-11T11:07:40.663921Z","iopub.status.idle":"2022-07-11T11:07:40.694022Z","shell.execute_reply.started":"2022-07-11T11:07:40.663877Z","shell.execute_reply":"2022-07-11T11:07:40.693215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = df.drop('Survived', axis=1).values\ny_train = df['Survived'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.695584Z","iopub.execute_input":"2022-07-11T11:07:40.696009Z","iopub.status.idle":"2022-07-11T11:07:40.708566Z","shell.execute_reply.started":"2022-07-11T11:07:40.695955Z","shell.execute_reply":"2022-07-11T11:07:40.706434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.710831Z","iopub.execute_input":"2022-07-11T11:07:40.711262Z","iopub.status.idle":"2022-07-11T11:07:40.743833Z","shell.execute_reply.started":"2022-07-11T11:07:40.711218Z","shell.execute_reply":"2022-07-11T11:07:40.742659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.745156Z","iopub.execute_input":"2022-07-11T11:07:40.745585Z","iopub.status.idle":"2022-07-11T11:07:40.761440Z","shell.execute_reply.started":"2022-07-11T11:07:40.745545Z","shell.execute_reply":"2022-07-11T11:07:40.760184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Fare'] = df_test['Fare'].fillna(value=df_test['Fare'].mean())\n\nfor i in range(0,len(df_test['Cabin'])):\n    if not pd.isna(df_test['Cabin'][i]):\n        df_test['Cabin'][i] = df_test['Cabin'][i][0]\n    else:\n        pass\n\nfrom sklearn.preprocessing import LabelEncoder\ndf_2 = df_test.copy()\n\noriginal = df_2\nmask = df_2['Cabin'].isnull()\n\ndf_2 = df_2.astype(str).apply(LabelEncoder().fit_transform)\ndf_test['Cabin'] = df_2.where(~mask, original)['Cabin']\n\nfor i in range(0, len(df_test['Cabin'])):\n    if pd.isna(df_test['Cabin'][i]):\n        df_test['Cabin'][i] = regressor.predict([[df_test['Fare'][i]]]).round().astype(int)[0]\n    else:\n        pass\n\nfrom sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\ndf_test['Sex'] = le.fit_transform(df_test['Sex'])\ndf_test['Embarked'] = le.fit_transform(df_test['Embarked'])\n\ndf_test['Age'] = df_test['Age'].fillna(value=df_test['Age'].mean())\ndf_test['Age'] = df_test['Age'].astype('int')\n","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-11T11:07:40.763247Z","iopub.execute_input":"2022-07-11T11:07:40.764395Z","iopub.status.idle":"2022-07-11T11:07:40.991676Z","shell.execute_reply.started":"2022-07-11T11:07:40.764345Z","shell.execute_reply":"2022-07-11T11:07:40.990513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:40.993319Z","iopub.execute_input":"2022-07-11T11:07:40.993753Z","iopub.status.idle":"2022-07-11T11:07:41.010414Z","shell.execute_reply.started":"2022-07-11T11:07:40.993710Z","shell.execute_reply":"2022-07-11T11:07:41.009320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['Cabin'] = df_test['Cabin'].astype(int)\ndf['Cabin'] = df['Cabin'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.012290Z","iopub.execute_input":"2022-07-11T11:07:41.012709Z","iopub.status.idle":"2022-07-11T11:07:41.021660Z","shell.execute_reply.started":"2022-07-11T11:07:41.012667Z","shell.execute_reply":"2022-07-11T11:07:41.020790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = df_test.drop(['Name','Ticket'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.024616Z","iopub.execute_input":"2022-07-11T11:07:41.025978Z","iopub.status.idle":"2022-07-11T11:07:41.035821Z","shell.execute_reply.started":"2022-07-11T11:07:41.025928Z","shell.execute_reply":"2022-07-11T11:07:41.034828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = pd.DataFrame()\npred['PassengerId'] = df_test['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.037182Z","iopub.execute_input":"2022-07-11T11:07:41.037767Z","iopub.status.idle":"2022-07-11T11:07:41.048939Z","shell.execute_reply.started":"2022-07-11T11:07:41.037736Z","shell.execute_reply":"2022-07-11T11:07:41.047860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate(model):\n    model.fit(X_train,y_train)\n    print('nom du modèle : ',model)\n    pred['Survived'] = model.predict(X_test)\n    # pred.to_csv('Subbmission_9.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.050723Z","iopub.execute_input":"2022-07-11T11:07:41.051185Z","iopub.status.idle":"2022-07-11T11:07:41.059158Z","shell.execute_reply.started":"2022-07-11T11:07:41.051139Z","shell.execute_reply":"2022-07-11T11:07:41.058014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from hyperopt import STATUS_OK, Trials, fmin, hp, tpe\nspace={'max_depth': hp.quniform(\"max_depth\", 3, 18, 1),\n        'gamma': hp.uniform ('gamma', 1,9),\n        'reg_alpha' : hp.quniform('reg_alpha', 40,180,1),\n        'reg_lambda' : hp.uniform('reg_lambda', 0,1),\n        'colsample_bytree' : hp.uniform('colsample_bytree', 0.5,1),\n        'min_child_weight' : hp.quniform('min_child_weight', 0, 10, 1),\n        'n_estimators': 180,\n        'seed': 0\n    }","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.060990Z","iopub.execute_input":"2022-07-11T11:07:41.061449Z","iopub.status.idle":"2022-07-11T11:07:41.430512Z","shell.execute_reply.started":"2022-07-11T11:07:41.061407Z","shell.execute_reply":"2022-07-11T11:07:41.429385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\n\nm1=LogisticRegression()\nm2=SVC()\nm3=DecisionTreeClassifier(max_depth=6)\nm4=RandomForestClassifier(max_samples=0.9)\nm5=KNeighborsClassifier(n_neighbors=5)\nm6=XGBClassifier(\n                    n_estimators =space['n_estimators'], max_depth = space['max_depth'], gamma = space['gamma'],\n                    reg_alpha = space['reg_alpha'],min_child_weight=space['min_child_weight'],\n                    colsample_bytree=space['colsample_bytree'])\nm7=LGBMClassifier(max_depth=6, random_state=314, silent=True, metric='None', n_jobs=6)\n\nmodels=[m7]\n\nfor model in models:\n    evaluate(model);","metadata":{"execution":{"iopub.status.busy":"2022-07-11T11:07:41.432036Z","iopub.execute_input":"2022-07-11T11:07:41.432467Z","iopub.status.idle":"2022-07-11T11:07:42.410131Z","shell.execute_reply.started":"2022-07-11T11:07:41.432425Z","shell.execute_reply":"2022-07-11T11:07:42.408937Z"},"trusted":true},"execution_count":null,"outputs":[]}]}