{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:29:12.085761Z","iopub.execute_input":"2024-04-28T08:29:12.086237Z","iopub.status.idle":"2024-04-28T08:29:12.092688Z","shell.execute_reply.started":"2024-04-28T08:29:12.086201Z","shell.execute_reply":"2024-04-28T08:29:12.091115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyspark","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:29:36.438322Z","iopub.execute_input":"2024-04-28T08:29:36.439156Z","iopub.status.idle":"2024-04-28T08:29:36.445521Z","shell.execute_reply.started":"2024-04-28T08:29:36.439115Z","shell.execute_reply":"2024-04-28T08:29:36.443959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql import SparkSession\nspark = SparkSession.builder.getOrCreate()","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:29:40.689569Z","iopub.execute_input":"2024-04-28T08:29:40.689996Z","iopub.status.idle":"2024-04-28T08:29:40.702795Z","shell.execute_reply.started":"2024-04-28T08:29:40.689962Z","shell.execute_reply":"2024-04-28T08:29:40.701767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.sql.functions import to_date, col\nfrom pyspark.sql.functions import to_date, dayofweek\nfrom pyspark.sql.functions import mean, stddev\nfrom pyspark.sql.functions import round\nfrom pyspark.sql.functions import log\nfrom pyspark.ml.feature import Binarizer\nfrom pyspark.ml.feature import Bucketizer\nfrom pyspark.ml.feature import VectorAssembler\nfrom pyspark.sql.types import StringType, DateType\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:30:10.954473Z","iopub.execute_input":"2024-04-28T08:30:10.954984Z","iopub.status.idle":"2024-04-28T08:30:12.207935Z","shell.execute_reply.started":"2024-04-28T08:30:10.954949Z","shell.execute_reply":"2024-04-28T08:30:12.206555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **AMERICAN EXPRESS - DEFAULT PREDICTION**","metadata":{}},{"cell_type":"code","source":"df = spark.read.csv('/kaggle/input/amex-default-prediction/sample_submission.csv')\ntest_dt = spark.read.csv('/kaggle/input/amex-default-prediction/test_data.csv')\ntrain_dt = spark.read.csv('/kaggle/input/amex-default-prediction/train_data.csv')\ntrain_lab = spark.read.csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:43:16.791919Z","iopub.execute_input":"2024-04-28T08:43:16.792494Z","iopub.status.idle":"2024-04-28T08:43:18.431576Z","shell.execute_reply.started":"2024-04-28T08:43:16.792457Z","shell.execute_reply":"2024-04-28T08:43:18.430218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_lab.show(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T08:57:16.391108Z","iopub.execute_input":"2024-04-28T08:57:16.392586Z","iopub.status.idle":"2024-04-28T08:57:16.543631Z","shell.execute_reply.started":"2024-04-28T08:57:16.392545Z","shell.execute_reply":"2024-04-28T08:57:16.542471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt.printSchema()","metadata":{"execution":{"iopub.status.busy":"2024-04-28T09:09:54.043911Z","iopub.execute_input":"2024-04-28T09:09:54.044828Z","iopub.status.idle":"2024-04-28T09:09:54.054229Z","shell.execute_reply.started":"2024-04-28T09:09:54.044789Z","shell.execute_reply":"2024-04-28T09:09:54.05268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Clean Data**","metadata":{}},{"cell_type":"code","source":"test_dt_first_row = test_dt.first()\ntest_dt_first_row","metadata":{"execution":{"iopub.status.busy":"2024-04-28T09:10:28.083237Z","iopub.execute_input":"2024-04-28T09:10:28.084708Z","iopub.status.idle":"2024-04-28T09:10:28.318392Z","shell.execute_reply.started":"2024-04-28T09:10:28.084655Z","shell.execute_reply":"2024-04-28T09:10:28.302345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt = test_dt.filter(df.columns != test_dt_first_row).toDF(*header)","metadata":{"execution":{"iopub.status.busy":"2024-04-28T09:09:15.163344Z","iopub.execute_input":"2024-04-28T09:09:15.1638Z","iopub.status.idle":"2024-04-28T09:09:15.215949Z","shell.execute_reply.started":"2024-04-28T09:09:15.163768Z","shell.execute_reply":"2024-04-28T09:09:15.214367Z"},"trusted":true},"execution_count":null,"outputs":[]}]}