{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.tree import DecisionTreeRegressor\nfrom sklearn.metrics import mean_absolute_error\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier\nfrom xgboost import XGBRegressor\n\n# deep learing\nimport matplotlib.pyplot as plt\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nimport tensorflow as tf\n\n\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import make_column_transformer\nfrom sklearn.model_selection import GroupShuffleSplit\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom tensorflow.keras import callbacks\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.compose import make_column_transformer\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import make_column_transformer, make_column_selector\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.compose import make_column_transformer\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T23:14:21.347973Z","iopub.execute_input":"2022-08-10T23:14:21.348353Z","iopub.status.idle":"2022-08-10T23:14:21.365645Z","shell.execute_reply.started":"2022-08-10T23:14:21.348323Z","shell.execute_reply":"2022-08-10T23:14:21.364310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data\ntrain_data = pd.read_csv('../input/titanic/train.csv')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.468609Z","iopub.execute_input":"2022-08-10T23:14:21.469360Z","iopub.status.idle":"2022-08-10T23:14:21.492384Z","shell.execute_reply.started":"2022-08-10T23:14:21.469320Z","shell.execute_reply":"2022-08-10T23:14:21.491532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data\nX_test = pd.read_csv('../input/titanic/test.csv') \nX_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.592750Z","iopub.execute_input":"2022-08-10T23:14:21.593460Z","iopub.status.idle":"2022-08-10T23:14:21.628244Z","shell.execute_reply.started":"2022-08-10T23:14:21.593413Z","shell.execute_reply":"2022-08-10T23:14:21.627013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read data\nval_data = pd.read_csv('../input/titanic/gender_submission.csv')\nval_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.707652Z","iopub.execute_input":"2022-08-10T23:14:21.708528Z","iopub.status.idle":"2022-08-10T23:14:21.729861Z","shell.execute_reply.started":"2022-08-10T23:14:21.708487Z","shell.execute_reply":"2022-08-10T23:14:21.728648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get how colums have missing value \ncols_with_missing = [col for col in train_data.columns\n                     if train_data[col].isnull().any()]\nprint(cols_with_missing)\n\n#get how missing value in each colum\nmissing_val_count_by_column = (train_data.isnull().sum())\nprint(missing_val_count_by_column)\n\n# get how catgories colums\ns = (train_data.dtypes == 'object')\nobject_cols = list(s[s].index)\nprint(object_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.775430Z","iopub.execute_input":"2022-08-10T23:14:21.775925Z","iopub.status.idle":"2022-08-10T23:14:21.794967Z","shell.execute_reply.started":"2022-08-10T23:14:21.775875Z","shell.execute_reply":"2022-08-10T23:14:21.793966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i will solve catgories value by One-Hot Encoding if colums has 3 values will extend to 3 colums by values and in row \n# has this value write 1 else 0\nlow_cardinality_cols = [col for col in object_cols if train_data[col].nunique() < 10]\n\n# Apply one-hot encoder to each column with categorical data\nOH_encoder = OneHotEncoder(handle_unknown='ignore', sparse=False)\nOH_cols_train = pd.DataFrame(OH_encoder.fit_transform(train_data[low_cardinality_cols]))\n\n# One-hot encoding removed index; put it back\nOH_cols_train.index = train_data.index\n\n# Remove categorical columns (will replace with one-hot encoding)\nnum_X_train = train_data.drop(object_cols, axis=1)\n\n# Add one-hot encoded columns to numerical features\nOH_X_train = pd.concat([num_X_train, OH_cols_train], axis=1)\nOH_X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.848398Z","iopub.execute_input":"2022-08-10T23:14:21.849147Z","iopub.status.idle":"2022-08-10T23:14:21.879364Z","shell.execute_reply.started":"2022-08-10T23:14:21.849110Z","shell.execute_reply":"2022-08-10T23:14:21.878104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i will solve missing value nan by Extension To Imputation by give value and extend colum has value true if it missing\n# else it will be false \ncols_with_missing = [col for col in OH_X_train.columns\n                     if OH_X_train[col].isnull().any()]\n\nmy_imputer = SimpleImputer()\nimputed_X_train = pd.DataFrame(my_imputer.fit_transform(OH_X_train))\n\n# Imputation removed column names; put them back\nimputed_X_train.columns = OH_X_train.columns\nimputed_X_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.913515Z","iopub.execute_input":"2022-08-10T23:14:21.913952Z","iopub.status.idle":"2022-08-10T23:14:21.949523Z","shell.execute_reply.started":"2022-08-10T23:14:21.913883Z","shell.execute_reply":"2022-08-10T23:14:21.948533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check is we having missing value or catogirels..\ncols_with_missing = [col for col in imputed_X_train.columns\n                     if imputed_X_train[col].isnull().any()]\n\nprint(cols_with_missing)\nmissing_val_count_by_column = (imputed_X_train.isnull().sum())\nprint(missing_val_count_by_column)\n\ns = (imputed_X_train.dtypes == 'object')\nobject_cols = list(s[s].index)\nprint(object_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:21.981506Z","iopub.execute_input":"2022-08-10T23:14:21.982378Z","iopub.status.idle":"2022-08-10T23:14:21.996977Z","shell.execute_reply.started":"2022-08-10T23:14:21.982333Z","shell.execute_reply":"2022-08-10T23:14:21.996105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train_data = train_data.dropna(axis=0)\nfeature_names = ['PassengerId','Pclass','Age','SibSp','Parch','Fare']\nX = imputed_X_train[feature_names]\ny = imputed_X_train.Survived\nX.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:22.047666Z","iopub.execute_input":"2022-08-10T23:14:22.048131Z","iopub.status.idle":"2022-08-10T23:14:22.065721Z","shell.execute_reply.started":"2022-08-10T23:14:22.048091Z","shell.execute_reply":"2022-08-10T23:14:22.064513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i divide data to training and validation \ntrain_X, val_X, train_y, val_y = train_test_split(X, y, random_state = 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:22.110660Z","iopub.execute_input":"2022-08-10T23:14:22.111130Z","iopub.status.idle":"2022-08-10T23:14:22.118614Z","shell.execute_reply.started":"2022-08-10T23:14:22.111089Z","shell.execute_reply":"2022-08-10T23:14:22.117581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i try model DecisionTreeRegressor\niowa_model =  DecisionTreeRegressor(random_state=1)\niowa_model .fit(train_X, train_y)\n# get prediction from validation x \nval_predictions = iowa_model.predict(val_X)\n#print(val_predictions)\n# get mean absolute error \nval_mae =mean_absolute_error(val_y, val_predictions)\nprint(val_mae)\nscore=iowa_model.score(val_X,val_y)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:22.223253Z","iopub.execute_input":"2022-08-10T23:14:22.223646Z","iopub.status.idle":"2022-08-10T23:14:22.240820Z","shell.execute_reply.started":"2022-08-10T23:14:22.223615Z","shell.execute_reply":"2022-08-10T23:14:22.239683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i try model RandomForestRegressor\nrf_model = XGBRegressor(random_state=0)\nrf_model .fit(train_X, train_y)\n# get prediction from validation x \nval_predictions = rf_model.predict(val_X)\n#print(val_predictions)\n# get mean absolute error \nval_mae =mean_absolute_error(val_y, val_predictions)\nprint(val_mae)\nscore=rf_model.score(val_X,val_y)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:22.313542Z","iopub.execute_input":"2022-08-10T23:14:22.314749Z","iopub.status.idle":"2022-08-10T23:14:22.655193Z","shell.execute_reply.started":"2022-08-10T23:14:22.314708Z","shell.execute_reply":"2022-08-10T23:14:22.654206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i try model RandomForestRegressor\nrf_model = RandomForestRegressor(random_state=1)\nrf_model .fit(train_X, train_y)\n# get prediction from validation x \nval_predictions = rf_model.predict(val_X)\n#print(val_predictions)\n# get mean absolute error \nval_mae =mean_absolute_error(val_y, val_predictions)\nprint(val_mae)\nscore=rf_model.score(val_X,val_y)\nprint(score)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:22.659632Z","iopub.execute_input":"2022-08-10T23:14:22.662371Z","iopub.status.idle":"2022-08-10T23:14:22.927108Z","shell.execute_reply.started":"2022-08-10T23:14:22.662324Z","shell.execute_reply":"2022-08-10T23:14:22.925877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# i try model RandomForestClassifier\nrfc_model = RandomForestClassifier(n_estimators=100)\nrfc_model .fit(train_X, train_y)\n# get prediction from validation x \nval_predictions = rfc_model.predict(val_X)\n#print(val_predictions)\n# get mean absolute error \nval_mae =mean_absolute_error(val_y, val_predictions)\nprint(val_mae)\nscore=rf_model.score(val_X,val_y)\nprint(score)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:19:09.640244Z","iopub.execute_input":"2022-08-10T23:19:09.640697Z","iopub.status.idle":"2022-08-10T23:19:09.885681Z","shell.execute_reply.started":"2022-08-10T23:19:09.640656Z","shell.execute_reply":"2022-08-10T23:19:09.884546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i will divdie data again to do pipeline to improve training\nX_train_full, X_valid_full, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2,random_state=0)\ncategorical_cols = [cname for cname in X_train_full.columns if\n                    X_train_full[cname].nunique() < 10 and \n                    X_train_full[cname].dtype == \"object\"]\nnumerical_cols = [cname for cname in X_train_full.columns if \n                X_train_full[cname].dtype in ['int64', 'float64']]\nmy_cols = categorical_cols + numerical_cols\nnumerical_transformer = SimpleImputer(strategy='constant')\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:14:23.184172Z","iopub.execute_input":"2022-08-10T23:14:23.184507Z","iopub.status.idle":"2022-08-10T23:14:23.196972Z","shell.execute_reply.started":"2022-08-10T23:14:23.184477Z","shell.execute_reply":"2022-08-10T23:14:23.195945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i do model DecisionTreeRegressor with pipeline and do same steps like fit and predicition and mae\niowa_model =  DecisionTreeRegressor(random_state=1)\nmy_pipeline_rf_model = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', iowa_model)\n                             ])\nmy_pipeline_rf_model .fit(X_train_full, y_train)\nval_predictions = my_pipeline_rf_model.predict(X_valid_full)\n#print(val_predictions)\nval_mae =mean_absolute_error(y_valid, val_predictions)\nprint(val_mae)\nscore=my_pipeline_rf_model.score(X_valid_full,y_valid)\nprint(score)\npreds_test = my_pipeline_rf_model.predict(X_test)\noutput = pd.DataFrame({'Id': X_test.index,\n                       'SalePrice': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:15:07.261833Z","iopub.execute_input":"2022-08-10T23:15:07.262773Z","iopub.status.idle":"2022-08-10T23:15:07.295761Z","shell.execute_reply.started":"2022-08-10T23:15:07.262729Z","shell.execute_reply":"2022-08-10T23:15:07.294509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i do model XGBRegressor with pipeline and do same steps like fit and predicition and mae\nrfc_model = XGBRegressor(random_state=0)\nmy_pipeline_rf_model = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', rfc_model)\n                             ])\nmy_pipeline_rf_model .fit(X_train_full, y_train)\nval_predictions = my_pipeline_rf_model.predict(X_valid_full)\n#print(val_predictions)\nval_mae =mean_absolute_error(y_valid, val_predictions)\nprint(val_mae)\nscore=my_pipeline_rf_model.score(X_valid_full,y_valid)\nprint(score)\npreds_test = my_pipeline_rf_model.predict(X_test)\noutput = pd.DataFrame({'Id': X_test.index,\n                       'SalePrice': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:15:29.492779Z","iopub.execute_input":"2022-08-10T23:15:29.493228Z","iopub.status.idle":"2022-08-10T23:15:29.985580Z","shell.execute_reply.started":"2022-08-10T23:15:29.493194Z","shell.execute_reply":"2022-08-10T23:15:29.984485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i do model RandomForestRegressor with pipeline and do same steps like fit and predicition and mae\nrf_model = RandomForestRegressor(random_state=1)\nmy_pipeline_rf_model = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', rf_model)\n                             ])\nmy_pipeline_rf_model .fit(X_train_full, y_train)\nval_predictions = my_pipeline_rf_model.predict(X_valid_full)\n#print(val_predictions)\nval_mae =mean_absolute_error(y_valid, val_predictions)\nprint(val_mae)\nscore=my_pipeline_rf_model.score(X_valid_full,y_valid)\nprint(score)\npreds_test = my_pipeline_rf_model.predict(X_test)\noutput = pd.DataFrame({'Id': X_test.index,\n                       'SalePrice': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:15:42.526672Z","iopub.execute_input":"2022-08-10T23:15:42.527131Z","iopub.status.idle":"2022-08-10T23:15:42.823946Z","shell.execute_reply.started":"2022-08-10T23:15:42.527092Z","shell.execute_reply":"2022-08-10T23:15:42.822918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i do model RandomForestClassifier with pipeline and do same steps like fit and predicition and mae\nrfc_model = RandomForestClassifier(n_estimators=100)\nmy_pipeline_rf_model = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', rfc_model)\n                             ])\nmy_pipeline_rf_model .fit(X_train_full, y_train)\nval_predictions = my_pipeline_rf_model.predict(X_valid_full)\n#print(val_predictions)\nval_mae =mean_absolute_error(y_valid, val_predictions)\nprint(val_mae)\nscore=my_pipeline_rf_model.score(X_valid_full,y_valid)\nprint(score)\npreds_test = my_pipeline_rf_model.predict(X_test)\noutput = pd.DataFrame({'Id': X_test.index,\n                       'SalePrice': preds_test})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:15:54.376366Z","iopub.execute_input":"2022-08-10T23:15:54.376754Z","iopub.status.idle":"2022-08-10T23:15:54.670186Z","shell.execute_reply.started":"2022-08-10T23:15:54.376720Z","shell.execute_reply":"2022-08-10T23:15:54.668856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# here i do to do nerual network and deep learing\n# input nodes to first layer are 6\ninput_shape = [6]\n\n# here i make the model  8 layer relu and BatchNormalization and  dropout\nmodel = keras.Sequential([\n    layers.Dense(32, activation='relu', input_shape=input_shape),\n    layers.Dense(64, activation='relu'),\n    layers.Dropout(0.1),\n    layers.Dense(64, activation='relu'),\n    layers.BatchNormalization(),\n    layers.Dropout(0.1),\n    layers.Dense(1,activation='sigmoid'),\n    layers.BatchNormalization(),\n    \n])\n#when i change loss fro, mae to binary_crossentropy it make accuracy from 49 to 97\n#model.compile( optimizer='adam',loss='mae')\nmodel.compile( optimizer='adam',loss='binary_crossentropy')\n#fit data to train model \nmodel.fit(X,y,batch_size=128,epochs=50,verbose=0,)\n#evaluate model to get accuracy\naccuracy = model.evaluate(X, y)\nprint(accuracy)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T23:16:23.045010Z","iopub.execute_input":"2022-08-10T23:16:23.045786Z","iopub.status.idle":"2022-08-10T23:16:25.693156Z","shell.execute_reply.started":"2022-08-10T23:16:23.045738Z","shell.execute_reply":"2022-08-10T23:16:25.691662Z"},"trusted":true},"execution_count":null,"outputs":[]}]}