{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T17:20:43.043602Z","iopub.execute_input":"2022-07-18T17:20:43.044379Z","iopub.status.idle":"2022-07-18T17:20:43.075948Z","shell.execute_reply.started":"2022-07-18T17:20:43.043966Z","shell.execute_reply":"2022-07-18T17:20:43.075079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyspark","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:20:43.079368Z","iopub.execute_input":"2022-07-18T17:20:43.080287Z","iopub.status.idle":"2022-07-18T17:21:30.669400Z","shell.execute_reply.started":"2022-07-18T17:20:43.080249Z","shell.execute_reply":"2022-07-18T17:21:30.668089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyspark\nfrom pyspark.sql import SparkSession","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:30.671547Z","iopub.execute_input":"2022-07-18T17:21:30.671934Z","iopub.status.idle":"2022-07-18T17:21:30.735689Z","shell.execute_reply.started":"2022-07-18T17:21:30.671883Z","shell.execute_reply":"2022-07-18T17:21:30.734772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ***SPARK SESSION***","metadata":{}},{"cell_type":"code","source":"spark = SparkSession.builder.appName(\"First_spark\").getOrCreate()\nspark","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:30.737561Z","iopub.execute_input":"2022-07-18T17:21:30.738645Z","iopub.status.idle":"2022-07-18T17:21:38.318819Z","shell.execute_reply.started":"2022-07-18T17:21:30.738604Z","shell.execute_reply":"2022-07-18T17:21:38.317516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# > TRAINING DATA","metadata":{}},{"cell_type":"code","source":"df_train = spark.read.csv('../input/titanic/train.csv',header =True,inferSchema=True)\ndf_train.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:38.331879Z","iopub.execute_input":"2022-07-18T17:21:38.334042Z","iopub.status.idle":"2022-07-18T17:21:45.408024Z","shell.execute_reply.started":"2022-07-18T17:21:38.333996Z","shell.execute_reply":"2022-07-18T17:21:45.407001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.printSchema()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:45.409137Z","iopub.execute_input":"2022-07-18T17:21:45.409513Z","iopub.status.idle":"2022-07-18T17:21:45.432979Z","shell.execute_reply.started":"2022-07-18T17:21:45.409475Z","shell.execute_reply":"2022-07-18T17:21:45.432035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **EMPTY CELLS COUNT**","metadata":{}},{"cell_type":"code","source":"from pyspark.sql.functions import *\ndf_train.select([count(when(isnan(c) | col(c).isNull(), c)).alias(c) for c in df_train.columns]\n   ).show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:45.434000Z","iopub.execute_input":"2022-07-18T17:21:45.434332Z","iopub.status.idle":"2022-07-18T17:21:46.972300Z","shell.execute_reply.started":"2022-07-18T17:21:45.434297Z","shell.execute_reply":"2022-07-18T17:21:46.971287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.drop(\"Cabin\",\"Name\",\"Ticket\",\"PassengerId\")\ndf_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:46.973389Z","iopub.execute_input":"2022-07-18T17:21:46.973745Z","iopub.status.idle":"2022-07-18T17:21:47.005200Z","shell.execute_reply.started":"2022-07-18T17:21:46.973709Z","shell.execute_reply":"2022-07-18T17:21:47.004223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.feature import Imputer\nimput = Imputer(inputCol = 'Age', outputCol='Age_impute').setStrategy(\"mean\")\ndf_train=imput.fit(df_train).transform(df_train)\ndf_train.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:47.012457Z","iopub.execute_input":"2022-07-18T17:21:47.014707Z","iopub.status.idle":"2022-07-18T17:21:48.248744Z","shell.execute_reply.started":"2022-07-18T17:21:47.014672Z","shell.execute_reply":"2022-07-18T17:21:48.247690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.withColumn(\"Age_imputed\", round(df_train[\"Age_impute\"]))\ndf_train.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:48.249886Z","iopub.execute_input":"2022-07-18T17:21:48.251883Z","iopub.status.idle":"2022-07-18T17:21:48.507571Z","shell.execute_reply.started":"2022-07-18T17:21:48.251842Z","shell.execute_reply":"2022-07-18T17:21:48.506503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=df_train.na.drop(how='any',subset=['Embarked'])\ndf_train=df_train.drop('Age')\ndf_train.select([count(when(isnan(c) | col(c).isNull(), c)).alias(c) for c in df_train.columns]\n   ).show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:48.508686Z","iopub.execute_input":"2022-07-18T17:21:48.510801Z","iopub.status.idle":"2022-07-18T17:21:49.301138Z","shell.execute_reply.started":"2022-07-18T17:21:48.510762Z","shell.execute_reply":"2022-07-18T17:21:49.300145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyspark.ml.feature import StringIndexer\nindexer = StringIndexer(inputCols=['Sex','Embarked'], outputCols=['Sex_indx','Embarked_indx'])\ndf_train = indexer.fit(df_train).transform(df_train)\ndf_train.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:49.302222Z","iopub.execute_input":"2022-07-18T17:21:49.302578Z","iopub.status.idle":"2022-07-18T17:21:50.689686Z","shell.execute_reply.started":"2022-07-18T17:21:49.302540Z","shell.execute_reply":"2022-07-18T17:21:50.688674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = spark.read.csv('../input/titanic/test.csv',header=True,inferSchema=True)\ndf_test.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:50.690717Z","iopub.execute_input":"2022-07-18T17:21:50.691040Z","iopub.status.idle":"2022-07-18T17:21:51.236010Z","shell.execute_reply.started":"2022-07-18T17:21:50.691007Z","shell.execute_reply":"2022-07-18T17:21:51.233219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Making the same changes in the test data**","metadata":{}},{"cell_type":"code","source":"df_test=df_test.drop(\"Cabin\",\"Name\",\"Ticket\",\"PassengerId\")\ndf_test=imput.fit(df_test).transform(df_test)\ndf_test = df_test.withColumn(\"Age_imputed\", round(df_test[\"Age_impute\"]))\ndf_test=df_test.na.drop(how='any',subset=['Embarked'])\ndf_test=df_test.drop('Age')\ndf_test = indexer.fit(df_test).transform(df_test)\ndf_test.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:51.238148Z","iopub.execute_input":"2022-07-18T17:21:51.238514Z","iopub.status.idle":"2022-07-18T17:21:52.222384Z","shell.execute_reply.started":"2022-07-18T17:21:51.238475Z","shell.execute_reply":"2022-07-18T17:21:52.221365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:52.223528Z","iopub.execute_input":"2022-07-18T17:21:52.223876Z","iopub.status.idle":"2022-07-18T17:21:52.243342Z","shell.execute_reply.started":"2022-07-18T17:21:52.223838Z","shell.execute_reply":"2022-07-18T17:21:52.242369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Assembling the features**","metadata":{}},{"cell_type":"code","source":"from pyspark.ml.feature import VectorAssembler\nfeature_assembler = VectorAssembler(inputCols=['Pclass','SibSp','Parch','Fare',\n                                               'Age_imputed','Sex_indx','Embarked_indx'],\n                           outputCol = \"Features_assembled\")                       ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:52.244919Z","iopub.execute_input":"2022-07-18T17:21:52.245575Z","iopub.status.idle":"2022-07-18T17:21:52.263911Z","shell.execute_reply.started":"2022-07-18T17:21:52.245538Z","shell.execute_reply":"2022-07-18T17:21:52.262936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_input = feature_assembler.transform(df_train)\ntest_input = feature_assembler.transform(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:52.264998Z","iopub.execute_input":"2022-07-18T17:21:52.265365Z","iopub.status.idle":"2022-07-18T17:21:52.389832Z","shell.execute_reply.started":"2022-07-18T17:21:52.265329Z","shell.execute_reply":"2022-07-18T17:21:52.388593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_input.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:52.391064Z","iopub.execute_input":"2022-07-18T17:21:52.391437Z","iopub.status.idle":"2022-07-18T17:21:52.891416Z","shell.execute_reply.started":"2022-07-18T17:21:52.391374Z","shell.execute_reply":"2022-07-18T17:21:52.890434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train = train_input.select(\"Features_assembled\", \"Survived\")\nfinal_test = test_input.select(\"Features_assembled\")\nfinal_train.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:52.892498Z","iopub.execute_input":"2022-07-18T17:21:52.892842Z","iopub.status.idle":"2022-07-18T17:21:53.103503Z","shell.execute_reply.started":"2022-07-18T17:21:52.892806Z","shell.execute_reply":"2022-07-18T17:21:53.102362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# *Initializing and fitting the model with training data*","metadata":{}},{"cell_type":"code","source":"from pyspark.ml.classification import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:53.104562Z","iopub.execute_input":"2022-07-18T17:21:53.104915Z","iopub.status.idle":"2022-07-18T17:21:53.111496Z","shell.execute_reply.started":"2022-07-18T17:21:53.104878Z","shell.execute_reply":"2022-07-18T17:21:53.110045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg = LogisticRegression(featuresCol = \"Features_assembled\",labelCol=\"Survived\").fit(final_train)\n#log_reg = LogisticRegression(featuresCol='Features_assembled', labelCol='Survived')\n#log_reg.fit(final_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:53.114269Z","iopub.execute_input":"2022-07-18T17:21:53.115274Z","iopub.status.idle":"2022-07-18T17:21:56.944547Z","shell.execute_reply.started":"2022-07-18T17:21:53.115237Z","shell.execute_reply":"2022-07-18T17:21:56.943503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> ***I have to populate the survived column in test data beacause of the dimention mismatch with the training data** *","metadata":{}},{"cell_type":"code","source":"from pyspark.sql.functions import lit\nfinal_test = final_test.withColumn('Survived', lit(0))\nfinal_test.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:56.949636Z","iopub.execute_input":"2022-07-18T17:21:56.952081Z","iopub.status.idle":"2022-07-18T17:21:57.145599Z","shell.execute_reply.started":"2022-07-18T17:21:56.952041Z","shell.execute_reply":"2022-07-18T17:21:57.144506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_lin = log_reg.evaluate(final_test)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:57.146840Z","iopub.execute_input":"2022-07-18T17:21:57.147985Z","iopub.status.idle":"2022-07-18T17:21:57.439257Z","shell.execute_reply.started":"2022-07-18T17:21:57.147942Z","shell.execute_reply":"2022-07-18T17:21:57.438214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Here you can see the predictions on the prediction column and the probability***\n> **Ignore the Survived columns as i have to add that for the dimention missmatch with training data**","metadata":{}},{"cell_type":"code","source":"pred_lin.predictions.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T17:21:57.448343Z","iopub.execute_input":"2022-07-18T17:21:57.450776Z","iopub.status.idle":"2022-07-18T17:21:57.787515Z","shell.execute_reply.started":"2022-07-18T17:21:57.450734Z","shell.execute_reply":"2022-07-18T17:21:57.786479Z"},"trusted":true},"execution_count":null,"outputs":[]}]}