{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport seaborn as sns\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import train_test_split\n\n\nbg_color = 'white'\nktcolors = ['#d0384e', '#ee6445', '#fa9b58', '#fece7c', '#fff1a8', '#f4faad', '#d1ed9c', '#97d5a4', '#5cb7aa', '#3682ba']\nsns.set(rc={\"font.style\":\"normal\",\n            \"axes.facecolor\":bg_color,\n            \"figure.facecolor\":bg_color,\n            \"text.color\":\"black\",\n            \"xtick.color\":\"black\",\n            \"ytick.color\":\"black\",\n            \"axes.labelcolor\":\"black\",\n            \"axes.grid\":False,\n            'axes.labelsize':20,\n            'figure.figsize':(5.0, 5.0),\n            'xtick.labelsize':10,\n            'font.size':10,\n            'ytick.labelsize':10})\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T09:39:04.356982Z","iopub.execute_input":"2022-07-18T09:39:04.357403Z","iopub.status.idle":"2022-07-18T09:39:04.372554Z","shell.execute_reply.started":"2022-07-18T09:39:04.357347Z","shell.execute_reply":"2022-07-18T09:39:04.371026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/santander-customer-transaction-prediction/train.csv')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:03:25.174626Z","iopub.execute_input":"2022-07-18T09:03:25.174980Z","iopub.status.idle":"2022-07-18T09:03:34.769329Z","shell.execute_reply.started":"2022-07-18T09:03:25.174945Z","shell.execute_reply":"2022-07-18T09:03:34.768329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv('/kaggle/input/santander-customer-transaction-prediction/test.csv')\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:03:34.770618Z","iopub.execute_input":"2022-07-18T09:03:34.771024Z","iopub.status.idle":"2022-07-18T09:03:43.843433Z","shell.execute_reply.started":"2022-07-18T09:03:34.770987Z","shell.execute_reply":"2022-07-18T09:03:43.842182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:03:43.845815Z","iopub.execute_input":"2022-07-18T09:03:43.846190Z","iopub.status.idle":"2022-07-18T09:03:43.879047Z","shell.execute_reply.started":"2022-07-18T09:03:43.846156Z","shell.execute_reply":"2022-07-18T09:03:43.877493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:03:43.880931Z","iopub.execute_input":"2022-07-18T09:03:43.881384Z","iopub.status.idle":"2022-07-18T09:03:46.269556Z","shell.execute_reply.started":"2022-07-18T09:03:43.881319Z","shell.execute_reply":"2022-07-18T09:03:46.268507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Nulls - are there any?","metadata":{}},{"cell_type":"code","source":"df_train.isnull().sum().sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:03:46.270816Z","iopub.execute_input":"2022-07-18T09:03:46.271250Z","iopub.status.idle":"2022-07-18T09:03:46.377290Z","shell.execute_reply.started":"2022-07-18T09:03:46.271206Z","shell.execute_reply":"2022-07-18T09:03:46.376273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks like we don't have any empty values.","metadata":{}},{"cell_type":"markdown","source":"# ETA","metadata":{}},{"cell_type":"markdown","source":"My first notebooks runs gave me rather poor scores so now it's time to do some more exploratory data analysis.","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize=(40, 20))\nsns.heatmap(df_train.corr())\nax.set_title('Feature correlation matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:10:04.050710Z","iopub.execute_input":"2022-07-18T09:10:04.051878Z","iopub.status.idle":"2022-07-18T09:10:28.523946Z","shell.execute_reply.started":"2022-07-18T09:10:04.051812Z","shell.execute_reply":"2022-07-18T09:10:28.522776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data has no correlation. It hints me that there is no linear relationship among the features and target but there may be some non-linear relationship that will help the model.","metadata":{}},{"cell_type":"markdown","source":"## Target variable","metadata":{}},{"cell_type":"code","source":"x = df_train['target'].value_counts().values\nsns.barplot([0,1], x)\nplt.title('Target variable count')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:24:23.978986Z","iopub.execute_input":"2022-07-18T09:24:23.979447Z","iopub.status.idle":"2022-07-18T09:24:24.521196Z","shell.execute_reply.started":"2022-07-18T09:24:23.979410Z","shell.execute_reply":"2022-07-18T09:24:24.520039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- There is a class imbalance problem.\n- Let's undersample majority class.","metadata":{}},{"cell_type":"code","source":"from sklearn.utils import resample\n\ntarget_0 = df_train[df_train.target == 0]\ntarget_1 = df_train[df_train.target == 1]\n\ntarget_0_downsampled = resample(target_0, replace = False, n_samples = len(target_1), random_state = 13)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:32:32.111701Z","iopub.execute_input":"2022-07-18T09:32:32.112167Z","iopub.status.idle":"2022-07-18T09:32:32.337830Z","shell.execute_reply.started":"2022-07-18T09:32:32.112133Z","shell.execute_reply":"2022-07-18T09:32:32.336444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0_downsampled.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:32:45.105438Z","iopub.execute_input":"2022-07-18T09:32:45.108810Z","iopub.status.idle":"2022-07-18T09:32:45.115334Z","shell.execute_reply.started":"2022-07-18T09:32:45.108748Z","shell.execute_reply":"2022-07-18T09:32:45.114583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_0.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:33:03.450648Z","iopub.execute_input":"2022-07-18T09:33:03.451001Z","iopub.status.idle":"2022-07-18T09:33:03.457781Z","shell.execute_reply.started":"2022-07-18T09:33:03.450969Z","shell.execute_reply":"2022-07-18T09:33:03.456566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"downsampled = pd.concat([target_0_downsampled, target_1])\ndownsampled.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:34:21.174411Z","iopub.execute_input":"2022-07-18T09:34:21.175859Z","iopub.status.idle":"2022-07-18T09:34:21.222499Z","shell.execute_reply.started":"2022-07-18T09:34:21.175749Z","shell.execute_reply":"2022-07-18T09:34:21.221220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing X and y","metadata":{}},{"cell_type":"code","source":"X = downsampled.drop(['ID_code', 'target'], axis=1)\ny = downsampled['target']","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:37:15.631623Z","iopub.execute_input":"2022-07-18T09:37:15.632405Z","iopub.status.idle":"2022-07-18T09:37:15.661957Z","shell.execute_reply.started":"2022-07-18T09:37:15.632344Z","shell.execute_reply":"2022-07-18T09:37:15.660667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model training","metadata":{}},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.linear_model import LogisticRegression\n\nsteps = [(\"scaler\", StandardScaler()),\n         (\"logreg\", LogisticRegression())]\npipeline = Pipeline(steps)\n\n# Create the parameter space\nparameters = {\"logreg__C\": np.linspace(0.001, 1.0, 20)}\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, \n                                                    random_state=21)\n\n# Instantiate the grid search object\ncv = GridSearchCV(pipeline, param_grid=parameters)\n\n# Fit to the training data\ncv.fit(X_train, y_train)\nprint(cv.best_score_, \"\\n\", cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:39:08.950152Z","iopub.execute_input":"2022-07-18T09:39:08.951054Z","iopub.status.idle":"2022-07-18T09:39:35.571591Z","shell.execute_reply.started":"2022-07-18T09:39:08.951014Z","shell.execute_reply":"2022-07-18T09:39:35.568028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\n# classifier = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)\n# classifier.fit(X_train, y_train)\n\nsteps = [(\"scaler\", StandardScaler()),\n         (\"knn\", KNeighborsClassifier())]\npipeline = Pipeline(steps)\n\nparameters = {\"knn__n_neighbors\": range(1, 10)}\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, \n                                                    random_state=21)\n\n# Instantiate the grid search object\ncv = GridSearchCV(pipeline, param_grid=parameters)\n\n# Fit to the training data\ncv.fit(X_train, y_train)\nprint(cv.best_score_, \"\\n\", cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T09:48:49.795767Z","iopub.execute_input":"2022-07-18T09:48:49.796164Z","iopub.status.idle":"2022-07-18T09:52:29.255274Z","shell.execute_reply.started":"2022-07-18T09:48:49.796132Z","shell.execute_reply":"2022-07-18T09:52:29.253944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nsteps = [(\"scaler\", StandardScaler()),\n         (\"rf\", RandomForestClassifier(criterion = 'entropy'))]\npipeline = Pipeline(steps)\n\nparameters = {\"rf__n_estimators\": range(1, 15)}\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, \n                                                    random_state=21)\n\ncv = GridSearchCV(pipeline, param_grid=parameters)\n\n# Fit to the training data\ncv.fit(X_train, y_train)\nprint(cv.best_score_, \"\\n\", cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:37:56.580894Z","iopub.execute_input":"2022-07-18T10:37:56.581626Z","iopub.status.idle":"2022-07-18T10:44:11.187622Z","shell.execute_reply.started":"2022-07-18T10:37:56.581572Z","shell.execute_reply":"2022-07-18T10:44:11.186484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\nsteps = [(\"scaler\", StandardScaler()),\n         (\"gnb\", GaussianNB())]\npipeline = Pipeline(steps)\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, \n                                                    random_state=21)\n\npipeline.fit(X_train, y_train)\n\ny_pred = pipeline.predict(X_test)\n\nfrom sklearn.metrics import confusion_matrix, accuracy_score\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\naccuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:19:09.744988Z","iopub.execute_input":"2022-07-18T10:19:09.745437Z","iopub.status.idle":"2022-07-18T10:19:09.851199Z","shell.execute_reply.started":"2022-07-18T10:19:09.745400Z","shell.execute_reply":"2022-07-18T10:19:09.849400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Solution","metadata":{}},{"cell_type":"code","source":"X_target = df_test.drop(['ID_code'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:02:43.965207Z","iopub.execute_input":"2022-07-18T10:02:43.966001Z","iopub.status.idle":"2022-07-18T10:02:44.074158Z","shell.execute_reply.started":"2022-07-18T10:02:43.965955Z","shell.execute_reply":"2022-07-18T10:02:44.072661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps = [(\"scaler\", StandardScaler()),\n         (\"gnb\", GaussianNB())]\npipeline = Pipeline(steps)\n\npipeline.fit(X, y)\n\ny_pred = pipeline.predict(X_target)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:03:17.890394Z","iopub.execute_input":"2022-07-18T10:03:17.890812Z","iopub.status.idle":"2022-07-18T10:03:18.942120Z","shell.execute_reply.started":"2022-07-18T10:03:17.890778Z","shell.execute_reply":"2022-07-18T10:03:18.940594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output=pd.DataFrame({'ID_code': df_test.ID_code,\n                   'target': y_pred})\noutput.to_csv('submission.csv', index=False)\n\nprint(\"The submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T19:28:10.185726Z","iopub.execute_input":"2022-07-13T19:28:10.186121Z","iopub.status.idle":"2022-07-13T19:28:10.569840Z","shell.execute_reply.started":"2022-07-13T19:28:10.186090Z","shell.execute_reply":"2022-07-13T19:28:10.568342Z"},"trusted":true},"execution_count":null,"outputs":[]}]}