{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-05T18:36:06.921371Z","iopub.execute_input":"2023-07-05T18:36:06.922066Z","iopub.status.idle":"2023-07-05T18:36:06.931904Z","shell.execute_reply.started":"2023-07-05T18:36:06.922018Z","shell.execute_reply":"2023-07-05T18:36:06.930426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing Data","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/icr-identify-age-related-conditions/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:06.940502Z","iopub.execute_input":"2023-07-05T18:36:06.942011Z","iopub.status.idle":"2023-07-05T18:36:06.969354Z","shell.execute_reply.started":"2023-07-05T18:36:06.941947Z","shell.execute_reply":"2023-07-05T18:36:06.968052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Processing data","metadata":{}},{"cell_type":"markdown","source":"Replacing string with numbers","metadata":{}},{"cell_type":"code","source":"df['EJ'] = df['EJ'].replace('A',0)\ndf['EJ'] = df['EJ'].replace('B',1)\ndf = df.drop('Id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:06.973481Z","iopub.execute_input":"2023-07-05T18:36:06.975119Z","iopub.status.idle":"2023-07-05T18:36:06.987530Z","shell.execute_reply.started":"2023-07-05T18:36:06.975046Z","shell.execute_reply":"2023-07-05T18:36:06.985821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Splitting into training and testing","metadata":{}},{"cell_type":"code","source":"train = df.iloc[:560]\ntest = df.iloc[560:]","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:06.990386Z","iopub.execute_input":"2023-07-05T18:36:06.991363Z","iopub.status.idle":"2023-07-05T18:36:06.998065Z","shell.execute_reply.started":"2023-07-05T18:36:06.991309Z","shell.execute_reply":"2023-07-05T18:36:06.996605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Assigning x and y training and testing sets","metadata":{}},{"cell_type":"code","source":"train = train.dropna(axis=0)\ntest = test.dropna(axis=0)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:06.999644Z","iopub.execute_input":"2023-07-05T18:36:07.000386Z","iopub.status.idle":"2023-07-05T18:36:07.017642Z","shell.execute_reply.started":"2023-07-05T18:36:07.000348Z","shell.execute_reply":"2023-07-05T18:36:07.016335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = train.drop('Class', axis=1)\ny_train = train[\"Class\"]\n\n\nx_test = test.drop('Class', axis=1)\ny_test = test[\"Class\"]","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:07.021347Z","iopub.execute_input":"2023-07-05T18:36:07.022072Z","iopub.status.idle":"2023-07-05T18:36:07.033297Z","shell.execute_reply.started":"2023-07-05T18:36:07.022021Z","shell.execute_reply":"2023-07-05T18:36:07.032429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:07.035292Z","iopub.execute_input":"2023-07-05T18:36:07.036427Z","iopub.status.idle":"2023-07-05T18:36:07.046673Z","shell.execute_reply.started":"2023-07-05T18:36:07.036379Z","shell.execute_reply":"2023-07-05T18:36:07.045456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a model\n","metadata":{}},{"cell_type":"code","source":"model = LogisticRegression(max_iter=100000)\nmodel.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:07.050625Z","iopub.execute_input":"2023-07-05T18:36:07.051413Z","iopub.status.idle":"2023-07-05T18:36:11.066231Z","shell.execute_reply.started":"2023-07-05T18:36:07.051373Z","shell.execute_reply":"2023-07-05T18:36:11.064866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(x_test)\naccuracy = tf.reduce_mean(tf.cast(tf.equal(predictions, y_test), dtype=tf.float32))\nprint(\"Accuracy:\", accuracy.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.068689Z","iopub.execute_input":"2023-07-05T18:36:11.070028Z","iopub.status.idle":"2023-07-05T18:36:11.085207Z","shell.execute_reply.started":"2023-07-05T18:36:11.069972Z","shell.execute_reply":"2023-07-05T18:36:11.083900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_value = 13\ntest_object = x_train.iloc[[test_value]]\narr = test_object.values","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.086931Z","iopub.execute_input":"2023-07-05T18:36:11.088281Z","iopub.status.idle":"2023-07-05T18:36:11.098405Z","shell.execute_reply.started":"2023-07-05T18:36:11.088221Z","shell.execute_reply":"2023-07-05T18:36:11.096601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(test_object)\nprint(y_pred, y_train.iloc[[test_value]])","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.101968Z","iopub.execute_input":"2023-07-05T18:36:11.105961Z","iopub.status.idle":"2023-07-05T18:36:11.121106Z","shell.execute_reply.started":"2023-07-05T18:36:11.105896Z","shell.execute_reply":"2023-07-05T18:36:11.119825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\n\nbest_params = {'max_depth': 10, 'max_features': 'sqrt', 'min_samples_leaf': 1, 'min_samples_split': 2, 'n_estimators': 100}\nmodel_2 = RandomForestClassifier(max_depth=10, max_features=\"sqrt\", min_samples_leaf=1, min_samples_split=2, n_estimators=100)\nmodel_2.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.122802Z","iopub.execute_input":"2023-07-05T18:36:11.124176Z","iopub.status.idle":"2023-07-05T18:36:11.516759Z","shell.execute_reply.started":"2023-07-05T18:36:11.124115Z","shell.execute_reply":"2023-07-05T18:36:11.515457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_2.predict(x_test)\naccuracy = tf.reduce_mean(tf.cast(tf.equal(predictions, y_test), dtype=tf.float32))\nprint(\"Accuracy:\", accuracy.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.518517Z","iopub.execute_input":"2023-07-05T18:36:11.519869Z","iopub.status.idle":"2023-07-05T18:36:11.546188Z","shell.execute_reply.started":"2023-07-05T18:36:11.519812Z","shell.execute_reply":"2023-07-05T18:36:11.544833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TESTING","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n# DecisionTreeClassifier\n# LogisticRegression\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.naive_bayes import GaussianNB\n\nfrom sklearn.model_selection import GridSearchCV\n\nmodel_params = {\n    \"random_forest\": {\n        \"model\": RandomForestClassifier(),\n        \"params\": {\n            'n_estimators': [50, 100, 200],\n            'max_depth': [5, 10, 20],\n            'min_samples_split': [2, 5, 10],\n            'min_samples_leaf': [1, 2, 4],\n            'max_features': ['auto', 'sqrt', 'log2']\n        }\n    }\n}\n\n\n\nmodel_params_1 = {\n    \"random_forest\": {\n        \"model\": RandomForestClassifier(),\n        \"params\": {\n            'n_estimators': [50, 100, 200],\n            'max_depth': [5, 10, 20],\n            'min_samples_split': [2, 5, 10],\n            'min_samples_leaf': [1, 2, 4],\n            'max_features': ['auto', 'sqrt', 'log2']\n        }\n    },\n    \"decision_tree\": {\n        \"model\": DecisionTreeClassifier(),\n        \"params\": {\n            'criterion': ['gini', 'entropy'],\n            'splitter': ['best', 'random'],\n            'max_depth': [None, 1, 2, 3, 4, 5],\n            'min_samples_split': [2, 3, 4],\n            'min_samples_leaf': [1, 2, 3],\n            'min_weight_fraction_leaf': [0.0, 0.1],\n            'max_features': [None, 'auto', 'sqrt', 'log2'],\n        }\n    },\n    \"logistic_regresssion\": {\n        \"model\": LogisticRegression(),\n        \"params\": {\n            'penalty': ['l1', 'l2', 'elasticnet', 'none'],\n            'fit_intercept': [True, False],\n            'class_weight': [None, 'balanced'],\n            'solver': ['newton-cg', 'lbfgs', 'liblinear', 'sag', 'saga'],\n            'max_iter': [100, 200, 300],\n            'multi_class': ['auto', 'ovr', 'multinomial'],\n            'verbose': [0, 1, 2]\n        }\n    },\n    \"gradient_booster\": {\n        \"model\": GradientBoostingClassifier(),\n        \"params\": {\n        }\n    },\n    \"naive_bayes\": {\n        \"model\": GaussianNB(),\n        \"params\": {\n            'priors': [None, [0.1, 0.9], [0.2, 0.8], [0.3, 0.7], [0.4, 0.6], [0.5, 0.5], [0.6, 0.4], [0.7, 0.3], [0.8, 0.2], [0.9, 0.1]],\n        }\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.548124Z","iopub.execute_input":"2023-07-05T18:36:11.549310Z","iopub.status.idle":"2023-07-05T18:36:11.562334Z","shell.execute_reply.started":"2023-07-05T18:36:11.549264Z","shell.execute_reply":"2023-07-05T18:36:11.561044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores = []\n\n# for model_name, mp in model_params.items():\n#     clf =  GridSearchCV(mp['model'], mp['params'], cv=5, return_train_score=False)\n#     clf.fit(x_train, y_train)\n#     scores.append({\n#         'model': model_name,\n#         'best_score': clf.best_score_,\n#         'best_params': clf.best_params_\n#     })\n    \n# result_df_fr = pd.DataFrame(scores,columns=['model','best_score','best_params'])\n# result_df_fr","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.564343Z","iopub.execute_input":"2023-07-05T18:36:11.565255Z","iopub.status.idle":"2023-07-05T18:36:11.578169Z","shell.execute_reply.started":"2023-07-05T18:36:11.565211Z","shell.execute_reply":"2023-07-05T18:36:11.576873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scores = []\n# clf =  GridSearchCV(RandomForestClassifier(), {\n#             'n_estimators': [50, 100, 200],\n#             'max_depth': [5, 10, 20],\n#             'min_samples_split': [2, 5, 10],\n#             'min_samples_leaf': [1, 2, 4],\n#             'max_features': ['auto', 'sqrt', 'log2']\n#         }, cv=5, return_train_score=False)\n# clf.fit(x_train, y_train)\n\n# scores.append({\n#         'model': model_name,\n#         'best_score': clf.best_score_,\n#         'best_params': clf.best_params_\n# })\n\n# result_df_fr = pd.DataFrame(scores,columns=['model','best_score','best_params'])\n# result_df_fr","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.579989Z","iopub.execute_input":"2023-07-05T18:36:11.580961Z","iopub.status.idle":"2023-07-05T18:36:11.591930Z","shell.execute_reply.started":"2023-07-05T18:36:11.580878Z","shell.execute_reply":"2023-07-05T18:36:11.590587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# best_params = result_df_fr.loc[0, 'best_params']\n# print(best_params)\n# # {'max_depth': 10, 'max_features': 'sqrt', 'min_samples_leaf': 1, 'min_samples_split': 2, 'n_estimators': 100}","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.595846Z","iopub.execute_input":"2023-07-05T18:36:11.596449Z","iopub.status.idle":"2023-07-05T18:36:11.605758Z","shell.execute_reply.started":"2023-07-05T18:36:11.596397Z","shell.execute_reply":"2023-07-05T18:36:11.604373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predicting the Test dataset","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/icr-identify-age-related-conditions/test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.607513Z","iopub.execute_input":"2023-07-05T18:36:11.608052Z","iopub.status.idle":"2023-07-05T18:36:11.628619Z","shell.execute_reply.started":"2023-07-05T18:36:11.608000Z","shell.execute_reply":"2023-07-05T18:36:11.627121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replacing string values\ntest_df['EJ'] = df['EJ'].replace('A',0)\ntest_df['EJ'] = df['EJ'].replace('B',1)\n\n# Removing ID column because we need to fill NaN values\ncol_id = test_df.pop('Id')\n\n# Filling NaN values\ntest_df = test_df.apply(lambda x: x.fillna(x.mean()))\n\n\n# Re-Combining ID column\ntemp_df = pd.DataFrame({'Id': col_id})\nresult = pd.concat([temp_df, test_df], axis=1)\ntest_df = result","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.631077Z","iopub.execute_input":"2023-07-05T18:36:11.631669Z","iopub.status.idle":"2023-07-05T18:36:11.659739Z","shell.execute_reply.started":"2023-07-05T18:36:11.631613Z","shell.execute_reply":"2023-07-05T18:36:11.658493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Turning dataframe into one-row dataframe in an array\ndf_list = []\n\nfor index, row in test_df.iterrows():\n    new_df = pd.DataFrame(row).transpose()\n    df_list.append(new_df)\n    \nresult = pd.concat(df_list)\n\nprint(df_list[0])","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.661171Z","iopub.execute_input":"2023-07-05T18:36:11.661814Z","iopub.status.idle":"2023-07-05T18:36:11.684016Z","shell.execute_reply.started":"2023-07-05T18:36:11.661779Z","shell.execute_reply":"2023-07-05T18:36:11.682886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Id = []\nclass_0 = []\nclass_1 = []\n\n\nfor i in df_list:\n    person_data = i.drop('Id', axis=1)\n    person_name = i[\"Id\"]\n    \n    prediction = model_2.predict_proba(person_data)\n    \n\n    for e in person_name.values:\n        person_name = e\n    \n    \n    Id.append(person_name)\n    class_0.append(prediction[0][0])\n    class_1.append(prediction[0][1])","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.685634Z","iopub.execute_input":"2023-07-05T18:36:11.686304Z","iopub.status.idle":"2023-07-05T18:36:11.764234Z","shell.execute_reply.started":"2023-07-05T18:36:11.686264Z","shell.execute_reply":"2023-07-05T18:36:11.763063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df = df = pd.DataFrame({\n    'Id': Id,\n    'class_0': class_0,\n    'class_1': class_1\n})\n\n\nresult_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-07-05T18:36:11.765891Z","iopub.execute_input":"2023-07-05T18:36:11.766699Z","iopub.status.idle":"2023-07-05T18:36:11.774892Z","shell.execute_reply.started":"2023-07-05T18:36:11.766653Z","shell.execute_reply":"2023-07-05T18:36:11.773337Z"},"trusted":true},"execution_count":null,"outputs":[]}]}