{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Credits\n- https://www.geeksforgeeks.org/ml-logistic-regression-using-python/?ref=lbp\n- https://www.kaggle.com/code/odins0n/load-parquet-files-with-low-memory/\n- https://scikit-learn.org/stable/modules/impute.html","metadata":{}},{"cell_type":"markdown","source":"# Modules","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-23T19:59:09.715596Z","iopub.execute_input":"2022-07-23T19:59:09.716025Z","iopub.status.idle":"2022-07-23T19:59:11.220695Z","shell.execute_reply.started":"2022-07-23T19:59:09.715943Z","shell.execute_reply":"2022-07-23T19:59:11.219480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data","metadata":{}},{"cell_type":"code","source":"train = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\nlabels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T19:59:11.222772Z","iopub.execute_input":"2022-07-23T19:59:11.223123Z","iopub.status.idle":"2022-07-23T19:59:34.560734Z","shell.execute_reply.started":"2022-07-23T19:59:11.223092Z","shell.execute_reply":"2022-07-23T19:59:34.556323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.merge(train, labels, on='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T19:59:34.566444Z","iopub.execute_input":"2022-07-23T19:59:34.568152Z","iopub.status.idle":"2022-07-23T20:01:44.001223Z","shell.execute_reply.started":"2022-07-23T19:59:34.568089Z","shell.execute_reply":"2022-07-23T20:01:43.999935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = train.iloc[:, [2, 189]].values\ny = train.iloc[:, 190].values","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:44.004090Z","iopub.execute_input":"2022-07-23T20:01:44.004469Z","iopub.status.idle":"2022-07-23T20:01:46.004291Z","shell.execute_reply.started":"2022-07-23T20:01:44.004415Z","shell.execute_reply":"2022-07-23T20:01:46.002831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train the model","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(x, y, test_size = 0.25, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:46.006043Z","iopub.execute_input":"2022-07-23T20:01:46.006525Z","iopub.status.idle":"2022-07-23T20:01:47.837096Z","shell.execute_reply.started":"2022-07-23T20:01:46.006487Z","shell.execute_reply":"2022-07-23T20:01:47.835854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc_x = StandardScaler()\nxtrain = sc_x.fit_transform(X_train)\nxtest = sc_x.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:47.838510Z","iopub.execute_input":"2022-07-23T20:01:47.838781Z","iopub.status.idle":"2022-07-23T20:01:49.334385Z","shell.execute_reply.started":"2022-07-23T20:01:47.838753Z","shell.execute_reply":"2022-07-23T20:01:49.333245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imp = SimpleImputer(missing_values=np.nan, strategy='mean')\nimp.fit(xtrain)\nxtrain = imp.transform(xtrain)\n\nclassifier = LogisticRegression(random_state = 0)\nclassifier.fit(xtrain, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:49.335617Z","iopub.execute_input":"2022-07-23T20:01:49.335889Z","iopub.status.idle":"2022-07-23T20:01:54.911953Z","shell.execute_reply.started":"2022-07-23T20:01:49.335860Z","shell.execute_reply":"2022-07-23T20:01:54.911065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict","metadata":{}},{"cell_type":"code","source":"imp = SimpleImputer(missing_values=np.nan, strategy='mean')\nimp.fit(X_test)\nxtest = imp.transform(X_test)\ny_pred = classifier.predict(xtest)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:54.913321Z","iopub.execute_input":"2022-07-23T20:01:54.913843Z","iopub.status.idle":"2022-07-23T20:01:55.143944Z","shell.execute_reply.started":"2022-07-23T20:01:54.913806Z","shell.execute_reply":"2022-07-23T20:01:55.142858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"Accuracy : \", accuracy_score(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:55.145318Z","iopub.execute_input":"2022-07-23T20:01:55.145660Z","iopub.status.idle":"2022-07-23T20:01:55.262469Z","shell.execute_reply.started":"2022-07-23T20:01:55.145624Z","shell.execute_reply":"2022-07-23T20:01:55.260981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"markdown","source":"## Read Data","metadata":{}},{"cell_type":"code","source":"test = '../input/amex-default-prediction/test_data.csv'\ndf_temp = pd.DataFrame(columns=['customer_ID', 'prediction'])\ndf_temp.to_csv(\"submission.csv\", index=False, header=True)\nchunksize = 10 ** 6\nwith pd.read_csv(test, chunksize=chunksize) as reader:\n    for chunk in reader:\n        df_temp = pd.DataFrame()\n        x = chunk.iloc[:, [2, 189]].values\n        xtest = sc_x.transform(x)\n        imp = SimpleImputer(missing_values=np.nan, strategy='mean')\n        imp.fit(xtest)\n        xtest = imp.transform(xtest)\n        y_pred = classifier.predict(xtest)\n        df_temp['customer_ID'] = chunk.iloc[:, [0]]\n        df_temp['prediction'] = y_pred\n        df_temp.to_csv(\"submission.csv\", index=False, header=False, mode='a')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:01:55.266936Z","iopub.execute_input":"2022-07-23T20:01:55.267282Z","iopub.status.idle":"2022-07-23T20:18:29.265844Z","shell.execute_reply.started":"2022-07-23T20:01:55.267254Z","shell.execute_reply":"2022-07-23T20:18:29.264495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('./submission.csv')\nprint(len(submission))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:18:29.268592Z","iopub.execute_input":"2022-07-23T20:18:29.268950Z","iopub.status.idle":"2022-07-23T20:18:42.332640Z","shell.execute_reply.started":"2022-07-23T20:18:29.268920Z","shell.execute_reply":"2022-07-23T20:18:42.331296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = submission.groupby(['customer_ID']).max()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:42:28.109683Z","iopub.execute_input":"2022-07-23T20:42:28.110092Z","iopub.status.idle":"2022-07-23T20:42:32.016224Z","shell.execute_reply.started":"2022-07-23T20:42:28.110062Z","shell.execute_reply":"2022-07-23T20:42:32.015017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(df))\ndf.to_csv('submission.csv', index=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:42:38.214538Z","iopub.execute_input":"2022-07-23T20:42:38.215764Z","iopub.status.idle":"2022-07-23T20:42:41.545230Z","shell.execute_reply.started":"2022-07-23T20:42:38.215580Z","shell.execute_reply":"2022-07-23T20:42:41.544090Z"},"trusted":true},"execution_count":null,"outputs":[]}]}