{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Install Pyspark","metadata":{}},{"cell_type":"code","source":"!pip install -q pyspark","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:10:17.16947Z","iopub.execute_input":"2022-08-07T14:10:17.169924Z","iopub.status.idle":"2022-08-07T14:11:22.778026Z","shell.execute_reply.started":"2022-08-07T14:10:17.169811Z","shell.execute_reply":"2022-08-07T14:11:22.776233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import itertools\nimport multiprocessing\nimport re\nfrom IPython import display\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nimport pyspark.pandas as ps\nfrom pyspark import StorageLevel\nfrom pyspark.sql import SparkSession, types\nfrom pyspark.sql import functions as F\nfrom pyspark.ml.feature import OneHotEncoder, StringIndexer, VectorAssembler\nfrom pyspark.ml import Pipeline\nfrom pyspark.ml.classification import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:22.782827Z","iopub.execute_input":"2022-08-07T14:11:22.783215Z","iopub.status.idle":"2022-08-07T14:11:23.123977Z","shell.execute_reply.started":"2022-08-07T14:11:22.783139Z","shell.execute_reply":"2022-08-07T14:11:23.12274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating Spark Session","metadata":{}},{"cell_type":"code","source":"# SESSION PARAMETER\nCORES = multiprocessing.cpu_count()\nMAX_PARTITION_SIZE = \"134217728b\"","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:23.126052Z","iopub.execute_input":"2022-08-07T14:11:23.126765Z","iopub.status.idle":"2022-08-07T14:11:23.133873Z","shell.execute_reply.started":"2022-08-07T14:11:23.126725Z","shell.execute_reply":"2022-08-07T14:11:23.13233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spark = SparkSession.builder..appName(\"ML_spark\").getOrCreate()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spark = (SparkSession.builder.master(f\"local[{CORES}]\")\n                             .config(\"spark.memory.offHeap.enabled\", \"true\")\n                             .config(\"spark.memory.offHeap.size\",\"5g\")\n                             .config(\"spark.sql.shuffle.partitions\", CORES * 3)\n                             .config(\"spark.default.parallelism\", CORES * 3)\n                             .config(\"spark.sql.adaptive.advisoryPartitionSizeInBytes\", MAX_PARTITION_SIZE)\n                             .appName(\"ML_spark\")\n                             .getOrCreate())\nspark","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:23.13786Z","iopub.execute_input":"2022-08-07T14:11:23.138921Z","iopub.status.idle":"2022-08-07T14:11:32.177475Z","shell.execute_reply.started":"2022-08-07T14:11:23.138881Z","shell.execute_reply":"2022-08-07T14:11:32.175798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reading Data","metadata":{}},{"cell_type":"code","source":"train_path = \"../input/amex-pyspark-parquet/train_amex\"\ntest_path = \"../input/amex-pyspark-parquet/test_amex\"\nlabel_path = \"../input/amex-pyspark-parquet/label_amex\"","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:32.179525Z","iopub.execute_input":"2022-08-07T14:11:32.180629Z","iopub.status.idle":"2022-08-07T14:11:32.195246Z","shell.execute_reply.started":"2022-08-07T14:11:32.180491Z","shell.execute_reply":"2022-08-07T14:11:32.19239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = spark.read.parquet(train_path)\ntest_df = spark.read.parquet(test_path)\nlabel_df = spark.read.parquet(label_path)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:32.199815Z","iopub.execute_input":"2022-08-07T14:11:32.200782Z","iopub.status.idle":"2022-08-07T14:11:39.232835Z","shell.execute_reply.started":"2022-08-07T14:11:32.200741Z","shell.execute_reply":"2022-08-07T14:11:39.231365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#len of rows\ntrain_df.count()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:11:51.765628Z","iopub.execute_input":"2022-08-07T14:11:51.766013Z","iopub.status.idle":"2022-08-07T14:11:57.908268Z","shell.execute_reply.started":"2022-08-07T14:11:51.765982Z","shell.execute_reply":"2022-08-07T14:11:57.904937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#categorical columns\nfor col in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']:\n    print(col, train_df.select(col).dtypes)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:12:30.970129Z","iopub.execute_input":"2022-08-07T14:12:30.97058Z","iopub.status.idle":"2022-08-07T14:12:31.192806Z","shell.execute_reply.started":"2022-08-07T14:12:30.970549Z","shell.execute_reply":"2022-08-07T14:12:31.191517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_null_count(sql_df, colname):\n    count = (sql_df.select(colname)\n                   .filter(F.col(colname).isNull())\n                   .count())\n    return count\n\nfor col in ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']:\n    print(col, get_null_count(train_df,col))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:13:47.050738Z","iopub.execute_input":"2022-08-07T14:13:47.051149Z","iopub.status.idle":"2022-08-07T14:14:13.510665Z","shell.execute_reply.started":"2022-08-07T14:13:47.051119Z","shell.execute_reply":"2022-08-07T14:14:13.508794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filling the null value with 3 for D_66\ntrain_df = train_df.na.fill(value=\"3\",subset=[\"D_66\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:17:07.42187Z","iopub.execute_input":"2022-08-07T14:17:07.423137Z","iopub.status.idle":"2022-08-07T14:17:07.555534Z","shell.execute_reply.started":"2022-08-07T14:17:07.423103Z","shell.execute_reply":"2022-08-07T14:17:07.554142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_null_count(train_df,\"D_66\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:18:32.872443Z","iopub.execute_input":"2022-08-07T14:18:32.872853Z","iopub.status.idle":"2022-08-07T14:18:32.999082Z","shell.execute_reply.started":"2022-08-07T14:18:32.872821Z","shell.execute_reply":"2022-08-07T14:18:32.997699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. The target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:\n\nD_* = Delinquency variables S_* = Spend variables P_* = Payment variables B_* = Balance variables R_* = Risk variables with the following features being categorical:\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).\n\nNote that the negative class has been subsampled for this dataset at 5%, and thus receives a 20x weighting in the scoring metric.","metadata":{}},{"cell_type":"code","source":"varibales_dict = {\"S\":0,\"D\":0,\"P\":0,\"B\":0,\"R\":0}\nfor col in train_df.columns:\n    #varibales_dict[col[0]] = 0\n    if col[0] in varibales_dict.keys():\n        varibales_dict[col[0]] += 1\nvaribales_dict","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:19:07.400829Z","iopub.execute_input":"2022-08-07T14:19:07.402265Z","iopub.status.idle":"2022-08-07T14:19:07.429958Z","shell.execute_reply.started":"2022-08-07T14:19:07.402205Z","shell.execute_reply":"2022-08-07T14:19:07.428321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample Data","metadata":{}},{"cell_type":"code","source":"train_df.show(1, vertical=True)\nlabel_df.show(1, vertical=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:27:24.644646Z","iopub.execute_input":"2022-08-07T14:27:24.64517Z","iopub.status.idle":"2022-08-07T14:27:27.615305Z","shell.execute_reply.started":"2022-08-07T14:27:24.64513Z","shell.execute_reply":"2022-08-07T14:27:27.613996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## APPROACH 1 : Selecting last Transcation","metadata":{}},{"cell_type":"code","source":"def add_suffix(names, suffix):\n    return [name + suffix for name in names]","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:30:07.913348Z","iopub.execute_input":"2022-08-07T14:30:07.913839Z","iopub.status.idle":"2022-08-07T14:30:07.920616Z","shell.execute_reply.started":"2022-08-07T14:30:07.913793Z","shell.execute_reply":"2022-08-07T14:30:07.91925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Known Columns\ninfo_cols = ['customer_ID', 'S_2']\ntarget_cols = ['target']\ncat_cols = [\n    'B_30', 'B_38', \n    'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n\n# Define Numeric Columns\nexcluded = info_cols + cat_cols\nnum_cols = [col for col in train_df.columns if col not in excluded]\n\n# Define Feature Columns\nfeatures_cols =  cat_cols + num_cols\n\nprint(f\"Number of categoric cols: {len(cat_cols)}\")\nprint(f\"Number of numeric cols: {len(num_cols)}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:31:50.242844Z","iopub.execute_input":"2022-08-07T14:31:50.243445Z","iopub.status.idle":"2022-08-07T14:31:50.257252Z","shell.execute_reply.started":"2022-08-07T14:31:50.243415Z","shell.execute_reply":"2022-08-07T14:31:50.252877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Fill Missing Values\nThere are some columns in this dataframe that have two or more `null` value, our base strategies are:\n- Fill null in numeric columns with 0\n- Fill null in categoric columns with \"null\"","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:32:25.283775Z","iopub.execute_input":"2022-08-07T14:32:25.284234Z","iopub.status.idle":"2022-08-07T14:32:25.295544Z","shell.execute_reply.started":"2022-08-07T14:32:25.28416Z","shell.execute_reply":"2022-08-07T14:32:25.293338Z"}}},{"cell_type":"code","source":"train_df = (train_df.fillna(0, subset=num_cols)\n                    .fillna(\"null\", subset=cat_cols))\n\ntest_df = (test_df.fillna(0, subset=num_cols)\n                  .fillna(\"null\", subset=cat_cols))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:32:50.597523Z","iopub.execute_input":"2022-08-07T14:32:50.597948Z","iopub.status.idle":"2022-08-07T14:32:51.447921Z","shell.execute_reply.started":"2022-08-07T14:32:50.597917Z","shell.execute_reply":"2022-08-07T14:32:51.446486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## String Indexing","metadata":{}},{"cell_type":"code","source":"# Create columns aliases\ncat_index_cols = add_suffix(cat_cols, \"_index\")\n\n# Fit StringIndexer\nindexers = StringIndexer(inputCols=cat_cols, outputCols=cat_index_cols)\nindexers_model = indexers.fit(train_df)\n\n# Transform to data\ntrain_df_indexed = indexers_model.transform(train_df)\ntest_df_indexed = indexers_model.transform(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:34:57.067728Z","iopub.execute_input":"2022-08-07T14:34:57.068169Z","iopub.status.idle":"2022-08-07T14:35:18.749904Z","shell.execute_reply.started":"2022-08-07T14:34:57.068134Z","shell.execute_reply":"2022-08-07T14:35:18.748109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:35:43.752237Z","iopub.execute_input":"2022-08-07T14:35:43.752683Z","iopub.status.idle":"2022-08-07T14:35:43.80342Z","shell.execute_reply.started":"2022-08-07T14:35:43.752652Z","shell.execute_reply":"2022-08-07T14:35:43.802032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df_indexed.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:35:24.742846Z","iopub.execute_input":"2022-08-07T14:35:24.743394Z","iopub.status.idle":"2022-08-07T14:35:24.757695Z","shell.execute_reply.started":"2022-08-07T14:35:24.743351Z","shell.execute_reply":"2022-08-07T14:35:24.756101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# See what columns the indexer handle\nindexers.getInputCols()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:39:07.740701Z","iopub.execute_input":"2022-08-07T14:39:07.74114Z","iopub.status.idle":"2022-08-07T14:39:07.75097Z","shell.execute_reply.started":"2022-08-07T14:39:07.741109Z","shell.execute_reply":"2022-08-07T14:39:07.749325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# See the indexed columns\ntrain_df_indexed.select(\"B_30_index\").show(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:39:14.128805Z","iopub.execute_input":"2022-08-07T14:39:14.129218Z","iopub.status.idle":"2022-08-07T14:39:14.468561Z","shell.execute_reply.started":"2022-08-07T14:39:14.129186Z","shell.execute_reply":"2022-08-07T14:39:14.467031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## One Hot Encoding","metadata":{}},{"cell_type":"code","source":"# Create columns aliases\ncat_ohe_cols = add_suffix(cat_cols, \"_ohe\")\n\n# Fit OneHotEncoder\nohe = OneHotEncoder(inputCols=cat_index_cols, outputCols=cat_ohe_cols)\nohe_model = ohe.fit(train_df_indexed)\n\n# Transform to data\ntrain_df_ohed = ohe_model.transform(train_df_indexed)\ntest_df_ohed = ohe_model.transform(test_df_indexed)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:40:45.26966Z","iopub.execute_input":"2022-08-07T14:40:45.270243Z","iopub.status.idle":"2022-08-07T14:40:45.801427Z","shell.execute_reply.started":"2022-08-07T14:40:45.270192Z","shell.execute_reply":"2022-08-07T14:40:45.800182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_ohed.select(\"B_30_ohe\").show(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:40:54.036528Z","iopub.execute_input":"2022-08-07T14:40:54.037801Z","iopub.status.idle":"2022-08-07T14:40:54.600796Z","shell.execute_reply.started":"2022-08-07T14:40:54.037754Z","shell.execute_reply":"2022-08-07T14:40:54.599265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Approach 1 : Last transaction","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:41:40.027331Z","iopub.execute_input":"2022-08-07T14:41:40.027837Z","iopub.status.idle":"2022-08-07T14:41:40.037033Z","shell.execute_reply.started":"2022-08-07T14:41:40.027788Z","shell.execute_reply":"2022-08-07T14:41:40.034736Z"}}},{"cell_type":"markdown","source":"## Group Customer","metadata":{}},{"cell_type":"code","source":"train_df_ohed.select(\"S_2\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:43:45.836401Z","iopub.execute_input":"2022-08-07T14:43:45.836889Z","iopub.status.idle":"2022-08-07T14:43:45.873028Z","shell.execute_reply.started":"2022-08-07T14:43:45.836859Z","shell.execute_reply":"2022-08-07T14:43:45.871588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#if if there are any dates missing\nget_null_count(train_df_indexed,\"S_2\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:44:13.093181Z","iopub.execute_input":"2022-08-07T14:44:13.093612Z","iopub.status.idle":"2022-08-07T14:44:14.529851Z","shell.execute_reply.started":"2022-08-07T14:44:13.093582Z","shell.execute_reply":"2022-08-07T14:44:14.528592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql.window import Window\ntrain_new = train_df_ohed.withColumn(\"rn\", F.row_number()\n        .over(Window.partitionBy(\"customer_ID\")\n        .orderBy(F.col(\"S_2\").desc())))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T14:55:19.566728Z","iopub.execute_input":"2022-08-07T14:55:19.567343Z","iopub.status.idle":"2022-08-07T14:55:19.707144Z","shell.execute_reply.started":"2022-08-07T14:55:19.567312Z","shell.execute_reply":"2022-08-07T14:55:19.705778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new.select(\"rn\").show(15)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:03:40.524852Z","iopub.execute_input":"2022-08-07T15:03:40.525387Z","iopub.status.idle":"2022-08-07T15:03:45.36828Z","shell.execute_reply.started":"2022-08-07T15:03:40.525345Z","shell.execute_reply":"2022-08-07T15:03:45.366941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new2 = train_new.filter(F.col(\"rn\") == 1).drop(\"rn\")\n#train_new2.show(10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_new2.count()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:22:21.902144Z","iopub.execute_input":"2022-08-07T15:22:21.90267Z","iopub.status.idle":"2022-08-07T15:22:34.045081Z","shell.execute_reply.started":"2022-08-07T15:22:21.90261Z","shell.execute_reply":"2022-08-07T15:22:34.043314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql.window import Window\ntest_new = test_df_ohed.withColumn(\"rn\", F.row_number()\n        .over(Window.partitionBy(\"customer_ID\")\n        .orderBy(F.col(\"S_2\").desc())))\n\ntest_new2 = test_new.filter(F.col(\"rn\") == 1).drop(\"rn\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:57:39.136721Z","iopub.execute_input":"2022-08-07T15:57:39.137264Z","iopub.status.idle":"2022-08-07T15:57:39.291988Z","shell.execute_reply.started":"2022-08-07T15:57:39.137209Z","shell.execute_reply":"2022-08-07T15:57:39.290226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"train_joined_df = train_new2.join(F.broadcast(label_df), on=\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:28:41.163243Z","iopub.execute_input":"2022-08-07T15:28:41.163693Z","iopub.status.idle":"2022-08-07T15:28:41.289893Z","shell.execute_reply.started":"2022-08-07T15:28:41.16366Z","shell.execute_reply":"2022-08-07T15:28:41.287933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dim = len(train_joined_df.columns)\nprint(f\"Total features: {dim}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:29:01.400982Z","iopub.execute_input":"2022-08-07T15:29:01.401651Z","iopub.status.idle":"2022-08-07T15:29:01.418548Z","shell.execute_reply.started":"2022-08-07T15:29:01.401618Z","shell.execute_reply":"2022-08-07T15:29:01.417244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joined_df.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:36:53.259219Z","iopub.execute_input":"2022-08-07T15:36:53.259611Z","iopub.status.idle":"2022-08-07T15:36:53.273874Z","shell.execute_reply.started":"2022-08-07T15:36:53.25958Z","shell.execute_reply":"2022-08-07T15:36:53.271859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del_cols = [ 'B_30_index',\n 'B_38_index',\n 'D_114_index',\n 'D_116_index',\n 'D_117_index',\n 'D_120_index',\n 'D_126_index',\n 'D_63_index',\n 'D_64_index',\n 'D_66_index',\n 'D_68_index','B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:48:42.625364Z","iopub.execute_input":"2022-08-07T15:48:42.626344Z","iopub.status.idle":"2022-08-07T15:48:42.633122Z","shell.execute_reply.started":"2022-08-07T15:48:42.626303Z","shell.execute_reply":"2022-08-07T15:48:42.631529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joined_df_new = train_joined_df.drop(*del_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:50:00.254415Z","iopub.execute_input":"2022-08-07T15:50:00.254832Z","iopub.status.idle":"2022-08-07T15:50:00.305998Z","shell.execute_reply.started":"2022-08-07T15:50:00.254802Z","shell.execute_reply":"2022-08-07T15:50:00.304777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_joined_df_new = test_new2.drop(*del_cols)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:59:06.275102Z","iopub.execute_input":"2022-08-07T15:59:06.275579Z","iopub.status.idle":"2022-08-07T15:59:06.319894Z","shell.execute_reply.started":"2022-08-07T15:59:06.275549Z","shell.execute_reply":"2022-08-07T15:59:06.318649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_joined_df_new = train_joined_df_new.drop(\"S_2\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T16:02:15.784351Z","iopub.execute_input":"2022-08-07T16:02:15.785092Z","iopub.status.idle":"2022-08-07T16:02:15.822316Z","shell.execute_reply.started":"2022-08-07T16:02:15.785061Z","shell.execute_reply":"2022-08-07T16:02:15.820572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_joined_df_new = test_joined_df_new.drop(\"S_2\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T16:02:41.525774Z","iopub.execute_input":"2022-08-07T16:02:41.526274Z","iopub.status.idle":"2022-08-07T16:02:41.583912Z","shell.execute_reply.started":"2022-08-07T16:02:41.526226Z","shell.execute_reply":"2022-08-07T16:02:41.582227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_joined_df_new.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:59:18.172853Z","iopub.execute_input":"2022-08-07T15:59:18.173279Z","iopub.status.idle":"2022-08-07T15:59:18.183442Z","shell.execute_reply.started":"2022-08-07T15:59:18.17325Z","shell.execute_reply":"2022-08-07T15:59:18.181782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_joined_df_new.columns)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T15:59:33.605575Z","iopub.execute_input":"2022-08-07T15:59:33.605939Z","iopub.status.idle":"2022-08-07T15:59:33.662547Z","shell.execute_reply.started":"2022-08-07T15:59:33.605911Z","shell.execute_reply":"2022-08-07T15:59:33.660875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Assemble Vector","metadata":{}},{"cell_type":"code","source":"va = VectorAssembler(\n    inputCols=train_joined_df_new.drop(\"customer_ID\", \"target\").columns,\n    outputCol=\"features\",\n    handleInvalid=\"error\",\n)\n\ntrain_ready_df = (va.transform(train_joined_df_new)\n                    .select([\"customer_ID\", \"features\", \"target\"]))\n\ntest_ready_df = (va.transform(test_joined_df_new)\n                   .select([\"customer_ID\", \"features\"]))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T16:02:49.855589Z","iopub.execute_input":"2022-08-07T16:02:49.855978Z","iopub.status.idle":"2022-08-07T16:02:50.459966Z","shell.execute_reply.started":"2022-08-07T16:02:49.855947Z","shell.execute_reply":"2022-08-07T16:02:50.458701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"markdown","source":"## Train-Test split in Train Data","metadata":{}},{"cell_type":"code","source":"train,test = train_ready_df.randomSplit([0.7,0.3])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T16:03:24.029982Z","iopub.execute_input":"2022-08-07T16:03:24.030831Z","iopub.status.idle":"2022-08-07T16:03:24.084962Z","shell.execute_reply.started":"2022-08-07T16:03:24.030795Z","shell.execute_reply":"2022-08-07T16:03:24.083685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logres = LogisticRegression(featuresCol=\"features\", labelCol=\"target\")\nlogres_model = logres.fit(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T16:22:58.075118Z","iopub.execute_input":"2022-08-07T16:22:58.075654Z","iopub.status.idle":"2022-08-07T16:23:04.077153Z","shell.execute_reply.started":"2022-08-07T16:22:58.075604Z","shell.execute_reply":"2022-08-07T16:23:04.073327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spark.csv","metadata":{},"execution_count":null,"outputs":[]}]}