{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Exploratory Data analysis On Titanic</h1>\n","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.542101Z","iopub.execute_input":"2022-08-10T10:40:57.542474Z","iopub.status.idle":"2022-08-10T10:40:57.552325Z","shell.execute_reply.started":"2022-08-10T10:40:57.542444Z","shell.execute_reply":"2022-08-10T10:40:57.551034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Importing important libraries</h1>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.591704Z","iopub.execute_input":"2022-08-10T10:40:57.593375Z","iopub.status.idle":"2022-08-10T10:40:57.600247Z","shell.execute_reply.started":"2022-08-10T10:40:57.593333Z","shell.execute_reply":"2022-08-10T10:40:57.599272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Loading titanic dataset (train.csv)</h1>","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(\"/kaggle/input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.604016Z","iopub.execute_input":"2022-08-10T10:40:57.604487Z","iopub.status.idle":"2022-08-10T10:40:57.621900Z","shell.execute_reply.started":"2022-08-10T10:40:57.604442Z","shell.execute_reply":"2022-08-10T10:40:57.620806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Cleaning dataset</h1>","metadata":{}},{"cell_type":"code","source":"# preview of data \ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.623245Z","iopub.execute_input":"2022-08-10T10:40:57.624363Z","iopub.status.idle":"2022-08-10T10:40:57.643823Z","shell.execute_reply.started":"2022-08-10T10:40:57.624317Z","shell.execute_reply":"2022-08-10T10:40:57.642557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Describe the columns</h1>","metadata":{}},{"cell_type":"markdown","source":"* PassengerId => ID of the passenger\n* Survived    => For Survive in Accident value is: 1 died in the Accident value is: 0\n* Pclass      => Class of the passenger like railways 1st class,2nd class ,3rd class\n* Nmae        => Name of the passenger\n* Sex         => Sex of the passenger (male or female)\n* Age         => Age of the passenger\n* SibSp       =>  ---------  On the ship of the passenger\n* Parch => parent or child of the passenger on the ship\n* Ticket => Ticket number or something\n* Fare  => you can say ticket price\n* Cabin => cabin name the passenge is travling in\n* Embarked => passenger destination name start with ","metadata":{}},{"cell_type":"code","source":"# listing down the columns\ndf.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.648166Z","iopub.execute_input":"2022-08-10T10:40:57.648487Z","iopub.status.idle":"2022-08-10T10:40:57.656413Z","shell.execute_reply.started":"2022-08-10T10:40:57.648459Z","shell.execute_reply":"2022-08-10T10:40:57.655006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Categorical columns</h1>\n\n1. Survived\n2. Pclass\n3. sex \n4. sibsp\n5. parch \n6. embarjed\n\n<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Numerical columns</h1>\n\n1. age \n2. fare\n3. passangerid\n\n<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Mixed columns</h1>\n\n1. Name \n2. Ticked \n3. cabin","metadata":{}},{"cell_type":"code","source":"# for gatting the high lavel overview  by calling info mathed \ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.658154Z","iopub.execute_input":"2022-08-10T10:40:57.659147Z","iopub.status.idle":"2022-08-10T10:40:57.679784Z","shell.execute_reply.started":"2022-08-10T10:40:57.659106Z","shell.execute_reply":"2022-08-10T10:40:57.678763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how many missing values and which calunms\ndf.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.681633Z","iopub.execute_input":"2022-08-10T10:40:57.682391Z","iopub.status.idle":"2022-08-10T10:40:57.699891Z","shell.execute_reply.started":"2022-08-10T10:40:57.682355Z","shell.execute_reply":"2022-08-10T10:40:57.698554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Some canslusion till here</h1>\n\n1. missing values in calumns Cabin , Age and Embarked \n2. Cabin calumn have more then 70%  missing values so will have to drop the calumn\n3. Some calumns have inappopriate data types\n","metadata":{}},{"cell_type":"code","source":"# drop cabin calumn\ndf.drop(columns=[\"Cabin\"],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.701673Z","iopub.execute_input":"2022-08-10T10:40:57.702885Z","iopub.status.idle":"2022-08-10T10:40:57.713026Z","shell.execute_reply.started":"2022-08-10T10:40:57.702832Z","shell.execute_reply":"2022-08-10T10:40:57.712031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the missing values in age by mean()\ndf[\"Age\"].fillna(df[\"Age\"].mean(),inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.716320Z","iopub.execute_input":"2022-08-10T10:40:57.717091Z","iopub.status.idle":"2022-08-10T10:40:57.725721Z","shell.execute_reply.started":"2022-08-10T10:40:57.717044Z","shell.execute_reply":"2022-08-10T10:40:57.724888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the missing values in Embarked by mode()\ndf[\"Embarked\"].fillna(\"S\",inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.727244Z","iopub.execute_input":"2022-08-10T10:40:57.728226Z","iopub.status.idle":"2022-08-10T10:40:57.738807Z","shell.execute_reply.started":"2022-08-10T10:40:57.728193Z","shell.execute_reply":"2022-08-10T10:40:57.737876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"SibSp\"].value_counts()\n# you would niticed there five categary ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.740684Z","iopub.execute_input":"2022-08-10T10:40:57.741736Z","iopub.status.idle":"2022-08-10T10:40:57.755488Z","shell.execute_reply.started":"2022-08-10T10:40:57.741697Z","shell.execute_reply":"2022-08-10T10:40:57.753998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"Parch\"].value_counts()\n# the columns is also like th columns one  lets change the data of these columns","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.756987Z","iopub.execute_input":"2022-08-10T10:40:57.757329Z","iopub.status.idle":"2022-08-10T10:40:57.772192Z","shell.execute_reply.started":"2022-08-10T10:40:57.757283Z","shell.execute_reply":"2022-08-10T10:40:57.770893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Changing the data type of the following calumns </h1>\n\n1. Survived to category\n2. Pclass to category\n3. Sex to category\n4. age to intager\n5. Embarked to category\n\n* Following the <code>astype()</code> mathed \n","metadata":{}},{"cell_type":"code","source":"df[\"Survived\"]= df[\"Survived\"].astype(\"category\")\ndf[\"Pclass\"] = df[\"Pclass\"].astype(\"category\")\ndf[\"Sex\"] =df[\"Sex\"].astype(\"category\")\ndf[\"Age\"] =df[\"Age\"].astype(\"int\")\ndf[\"Embarked\"] = df[\"Embarked\"].astype(\"category\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.774064Z","iopub.execute_input":"2022-08-10T10:40:57.774789Z","iopub.status.idle":"2022-08-10T10:40:57.788691Z","shell.execute_reply.started":"2022-08-10T10:40:57.774745Z","shell.execute_reply":"2022-08-10T10:40:57.787732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# again we age calling  the info() mathed for check\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.791235Z","iopub.execute_input":"2022-08-10T10:40:57.791743Z","iopub.status.idle":"2022-08-10T10:40:57.814057Z","shell.execute_reply.started":"2022-08-10T10:40:57.791711Z","shell.execute_reply":"2022-08-10T10:40:57.812634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"you can see there are no missing values and also the data types has to changed","metadata":{}},{"cell_type":"code","source":"# five point using th describe mathed\ndf.describe()\n# for cheking the how spread the data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.815886Z","iopub.execute_input":"2022-08-10T10:40:57.816317Z","iopub.status.idle":"2022-08-10T10:40:57.854918Z","shell.execute_reply.started":"2022-08-10T10:40:57.816280Z","shell.execute_reply":"2022-08-10T10:40:57.854020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Univariate Analysis </h1>","metadata":{}},{"cell_type":"code","source":"# Univariate analysis\n# Let's start with th Survived columns\nsns.countplot(df[\"Survived\"])\ndeath_percent=round(df[\"Survived\"].value_counts().values[0]/891*100)\nprint(f\"Out of 891 {death_percent}% people died in the accident\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:57.856383Z","iopub.execute_input":"2022-08-10T10:40:57.856907Z","iopub.status.idle":"2022-08-10T10:40:58.036750Z","shell.execute_reply.started":"2022-08-10T10:40:57.856873Z","shell.execute_reply":"2022-08-10T10:40:58.035465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# people sapreted by pclass peercentage\nprint(df[\"Pclass\"].value_counts()/891*100)\nsns.countplot(df[\"Pclass\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:58.038411Z","iopub.execute_input":"2022-08-10T10:40:58.038757Z","iopub.status.idle":"2022-08-10T10:40:58.225139Z","shell.execute_reply.started":"2022-08-10T10:40:58.038726Z","shell.execute_reply":"2022-08-10T10:40:58.224219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Calclusion**: Pclass  3 is the most crowded class\n","metadata":{}},{"cell_type":"code","source":"# lets talk abut the gender\nprint(df[\"Sex\"].value_counts()/891*100)\nsns.countplot(df[\"Sex\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:58.226306Z","iopub.execute_input":"2022-08-10T10:40:58.227269Z","iopub.status.idle":"2022-08-10T10:40:58.353034Z","shell.execute_reply.started":"2022-08-10T10:40:58.227230Z","shell.execute_reply":"2022-08-10T10:40:58.351819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets talk about SibSp\nprint(df[\"SibSp\"].value_counts()/891*100)\nsns.countplot(df[\"SibSp\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:58.354729Z","iopub.execute_input":"2022-08-10T10:40:58.355452Z","iopub.status.idle":"2022-08-10T10:40:58.573163Z","shell.execute_reply.started":"2022-08-10T10:40:58.355410Z","shell.execute_reply":"2022-08-10T10:40:58.571668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parch column\nprint(df[\"Parch\"].value_counts()/891*100)\nsns.countplot(df[\"Parch\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:58.577025Z","iopub.execute_input":"2022-08-10T10:40:58.577386Z","iopub.status.idle":"2022-08-10T10:40:58.799567Z","shell.execute_reply.started":"2022-08-10T10:40:58.577355Z","shell.execute_reply":"2022-08-10T10:40:58.798038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Embarked\nprint(df[\"Embarked\"].value_counts()/891*100)\nsns.countplot(df[\"Embarked\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:58.801606Z","iopub.execute_input":"2022-08-10T10:40:58.802077Z","iopub.status.idle":"2022-08-10T10:40:59.242479Z","shell.execute_reply.started":"2022-08-10T10:40:58.802040Z","shell.execute_reply":"2022-08-10T10:40:59.240943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(df[\"Age\"],kde=True,alpha=.1)\n# let's ckeck out data is normaly disriputed\nprint(df[\"Age\"].skew()) # if skew is -2 ot 2 the data is normaly disitibuted\nprint(df[\"Age\"].kurt()) # if kurt is -7 to 7 then data is normaly distibued\n# by seeing to the graph you can say that most of the people's age betbeen 20 to 40 years","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:59.244514Z","iopub.execute_input":"2022-08-10T10:40:59.244865Z","iopub.status.idle":"2022-08-10T10:40:59.616510Z","shell.execute_reply.started":"2022-08-10T10:40:59.244833Z","shell.execute_reply":"2022-08-10T10:40:59.614791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(df[\"Age\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:59.618586Z","iopub.execute_input":"2022-08-10T10:40:59.618959Z","iopub.status.idle":"2022-08-10T10:40:59.802215Z","shell.execute_reply.started":"2022-08-10T10:40:59.618911Z","shell.execute_reply":"2022-08-10T10:40:59.800799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# just out of curiosity\n\nprint(\"people with the age in betbeen 60 to 70 are: \" + str(df[(df[\"Age\"]>60) & (df[\"Age\"]<70)].shape[0]))\nprint(\"people with the age in betbeen 60 to 70 are: \" + str(df[(df[\"Age\"]>70) & (df[\"Age\"]<75)].shape[0]))\nprint(\"people with the age in betbeen 60 to 70 are: \" + str(df[(df[\"Age\"]>75) ].shape[0]))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:59.803809Z","iopub.execute_input":"2022-08-10T10:40:59.804158Z","iopub.status.idle":"2022-08-10T10:40:59.823329Z","shell.execute_reply.started":"2022-08-10T10:40:59.804126Z","shell.execute_reply":"2022-08-10T10:40:59.820982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Canclusion:\n1. For prictical purposes age can be cosidered as normal distibution \n2. Deeper analysis required for outliers detection","metadata":{}},{"cell_type":"code","source":"sns.displot(df[\"Fare\"],kde=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:40:59.825047Z","iopub.execute_input":"2022-08-10T10:40:59.825852Z","iopub.status.idle":"2022-08-10T10:41:00.284495Z","shell.execute_reply.started":"2022-08-10T10:40:59.825726Z","shell.execute_reply":"2022-08-10T10:41:00.282991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df[\"Fare\"].skew())\nprint(df[\"Fare\"].kurt())\n# skew is 4.78 but for normal didtibution higher skew can be -2 to 2\n# kurt is  33,39 but for normal distiobution higher kurt can -7 to 7\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.286506Z","iopub.execute_input":"2022-08-10T10:41:00.286987Z","iopub.status.idle":"2022-08-10T10:41:00.294355Z","shell.execute_reply.started":"2022-08-10T10:41:00.286924Z","shell.execute_reply":"2022-08-10T10:41:00.292944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.boxplot(df[\"Fare\"])\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.296390Z","iopub.execute_input":"2022-08-10T10:41:00.297029Z","iopub.status.idle":"2022-08-10T10:41:00.467400Z","shell.execute_reply.started":"2022-08-10T10:41:00.296996Z","shell.execute_reply":"2022-08-10T10:41:00.465991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"people with the Fare in betbeen 200 to 300 are: \" + str(df[(df[\"Fare\"]>200) & (df[\"Fare\"]<300)].shape[0]))\nprint(\"people with the Fare in betbeen 200 to 300 are: \" + str(df[ (df[\"Fare\"]>300)].shape[0]))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.468968Z","iopub.execute_input":"2022-08-10T10:41:00.469420Z","iopub.status.idle":"2022-08-10T10:41:00.480027Z","shell.execute_reply.started":"2022-08-10T10:41:00.469378Z","shell.execute_reply":"2022-08-10T10:41:00.478491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Canclusion**:\n1. Highely skewed data a lot of people had cheaper ticket\n2. Outliers are there in the data","metadata":{}},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Let's start with multivariate analysis</h1>","metadata":{}},{"cell_type":"code","source":"# Survival with Pclss \nsns.countplot(df[\"Survived\"],hue=df[\"Pclass\"])\npd.crosstab(df[\"Pclass\"],df[\"Survived\"]).apply(lambda r: r/r.sum()*100,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.482246Z","iopub.execute_input":"2022-08-10T10:41:00.482778Z","iopub.status.idle":"2022-08-10T10:41:00.734696Z","shell.execute_reply.started":"2022-08-10T10:41:00.482732Z","shell.execute_reply":"2022-08-10T10:41:00.733215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Survival with sex \nsns.countplot(df[\"Survived\"],hue=df[\"Sex\"])\npd.crosstab(df[\"Sex\"],df[\"Survived\"]).apply(lambda r: r/r.sum()*100,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.736868Z","iopub.execute_input":"2022-08-10T10:41:00.737302Z","iopub.status.idle":"2022-08-10T10:41:00.958605Z","shell.execute_reply.started":"2022-08-10T10:41:00.737267Z","shell.execute_reply":"2022-08-10T10:41:00.957173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Survival with Embarked \nsns.countplot(df[\"Survived\"],hue=df[\"Embarked\"])\npd.crosstab(df[\"Embarked\"],df[\"Survived\"]).apply(lambda r: r/r.sum()*100,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:00.960381Z","iopub.execute_input":"2022-08-10T10:41:00.960767Z","iopub.status.idle":"2022-08-10T10:41:01.213348Z","shell.execute_reply.started":"2022-08-10T10:41:00.960733Z","shell.execute_reply":"2022-08-10T10:41:01.211929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Survival with sex \nplt.figure(figsize=(10,6))\nsns.histplot(df[df[\"Survived\"]==0][\"Age\"],color=\"red\",kde=True)\nsns.histplot(df[df[\"Survived\"]==1][\"Age\"],color=\"green\",kde=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:01.214716Z","iopub.execute_input":"2022-08-10T10:41:01.215049Z","iopub.status.idle":"2022-08-10T10:41:01.577780Z","shell.execute_reply.started":"2022-08-10T10:41:01.215018Z","shell.execute_reply":"2022-08-10T10:41:01.576372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Survival with sex \nplt.figure(figsize=(15,6))\nsns.histplot(df[df[\"Survived\"]==0][\"Fare\"],color=\"red\",kde=True)\nsns.histplot(df[df[\"Survived\"]==1][\"Fare\"],color=\"green\",kde=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:01.579167Z","iopub.execute_input":"2022-08-10T10:41:01.579480Z","iopub.status.idle":"2022-08-10T10:41:02.005224Z","shell.execute_reply.started":"2022-08-10T10:41:01.579450Z","shell.execute_reply":"2022-08-10T10:41:02.003273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfcr=df.corr()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.006805Z","iopub.execute_input":"2022-08-10T10:41:02.007248Z","iopub.status.idle":"2022-08-10T10:41:02.014779Z","shell.execute_reply.started":"2022-08-10T10:41:02.007213Z","shell.execute_reply":"2022-08-10T10:41:02.013408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfcr","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.016719Z","iopub.execute_input":"2022-08-10T10:41:02.017112Z","iopub.status.idle":"2022-08-10T10:41:02.035851Z","shell.execute_reply.started":"2022-08-10T10:41:02.017078Z","shell.execute_reply":"2022-08-10T10:41:02.034567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(dfcr,annot=True,cmap=\"coolwarm\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.038172Z","iopub.execute_input":"2022-08-10T10:41:02.038646Z","iopub.status.idle":"2022-08-10T10:41:02.368159Z","shell.execute_reply.started":"2022-08-10T10:41:02.038600Z","shell.execute_reply":"2022-08-10T10:41:02.366625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(6,6))\nsns.clustermap(dfcr,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.375208Z","iopub.execute_input":"2022-08-10T10:41:02.375579Z","iopub.status.idle":"2022-08-10T10:41:02.904502Z","shell.execute_reply.started":"2022-08-10T10:41:02.375546Z","shell.execute_reply":"2022-08-10T10:41:02.903224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Feature Engineering </h1>","metadata":{}},{"cell_type":"code","source":"df[\"Family_size\"] = df[\"SibSp\"] + df[\"Parch\"]\n# for new calumn","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.906542Z","iopub.execute_input":"2022-08-10T10:41:02.907030Z","iopub.status.idle":"2022-08-10T10:41:02.913550Z","shell.execute_reply.started":"2022-08-10T10:41:02.906986Z","shell.execute_reply":"2022-08-10T10:41:02.912598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.915221Z","iopub.execute_input":"2022-08-10T10:41:02.915579Z","iopub.status.idle":"2022-08-10T10:41:02.942603Z","shell.execute_reply.started":"2022-08-10T10:41:02.915546Z","shell.execute_reply":"2022-08-10T10:41:02.941503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def family_type(number):\n    if number ==0:\n        return \"Alone\"\n    elif number >0 and number<=4:\n        return \"Medium\"\n    else:\n        return \"Large\"","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.944052Z","iopub.execute_input":"2022-08-10T10:41:02.945071Z","iopub.status.idle":"2022-08-10T10:41:02.950919Z","shell.execute_reply.started":"2022-08-10T10:41:02.944929Z","shell.execute_reply":"2022-08-10T10:41:02.949720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"Family_type\"] = df[\"Family_size\"].apply(family_type)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.952193Z","iopub.execute_input":"2022-08-10T10:41:02.953348Z","iopub.status.idle":"2022-08-10T10:41:02.968563Z","shell.execute_reply.started":"2022-08-10T10:41:02.953307Z","shell.execute_reply":"2022-08-10T10:41:02.967024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:02.971079Z","iopub.execute_input":"2022-08-10T10:41:02.971579Z","iopub.status.idle":"2022-08-10T10:41:03.000156Z","shell.execute_reply.started":"2022-08-10T10:41:02.971542Z","shell.execute_reply":"2022-08-10T10:41:02.998804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping unwanted columns\ndf.drop(columns=[\"SibSp\",\"Parch\",\"Family_size\"],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.002699Z","iopub.execute_input":"2022-08-10T10:41:03.003198Z","iopub.status.idle":"2022-08-10T10:41:03.011216Z","shell.execute_reply.started":"2022-08-10T10:41:03.003160Z","shell.execute_reply":"2022-08-10T10:41:03.009772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.012697Z","iopub.execute_input":"2022-08-10T10:41:03.014183Z","iopub.status.idle":"2022-08-10T10:41:03.035847Z","shell.execute_reply.started":"2022-08-10T10:41:03.014136Z","shell.execute_reply":"2022-08-10T10:41:03.034706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df[\"Family_type\"],hue=df[\"Survived\"])\npd.crosstab(df[\"Family_type\"],df[\"Survived\"]).apply(lambda r: r/r.sum()*100,axis=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.037218Z","iopub.execute_input":"2022-08-10T10:41:03.038709Z","iopub.status.idle":"2022-08-10T10:41:03.220209Z","shell.execute_reply.started":"2022-08-10T10:41:03.038664Z","shell.execute_reply":"2022-08-10T10:41:03.219013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Decting outliers</h1>\n\n1. On numerical data\n<ul>\n<li>If you data following normal distribution then you knew 99.7% data layes in between -3std and +3std \n</li>\n<li>If you data not following normal distribution then plot <code>boxplot()</code> for Dected the outliers = Q1-1.5IQR and Q3+1.5IQR\n</li>\n</ul>\n2. On Categorical data\n<ul>\n<li>\nIf col is highely inballanced for eg: 10000 male and a 2 famale then elimemnate female\n</li>\n</ul>","metadata":{}},{"cell_type":"code","source":"# handling outliers in age (almost normal)\ndf=df[df[\"Age\"]<(df[\"Age\"].mean() +3*df[\"Age\"].std() )]\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.221748Z","iopub.execute_input":"2022-08-10T10:41:03.223402Z","iopub.status.idle":"2022-08-10T10:41:03.234540Z","shell.execute_reply.started":"2022-08-10T10:41:03.223356Z","shell.execute_reply":"2022-08-10T10:41:03.233144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# handling outliers in fare col\n# finding quitiles\nq1 = np.percentile(df[\"Fare\"],25)\nq3 = np.percentile(df[\"Fare\"],75)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.235688Z","iopub.execute_input":"2022-08-10T10:41:03.236367Z","iopub.status.idle":"2022-08-10T10:41:03.245441Z","shell.execute_reply.started":"2022-08-10T10:41:03.236330Z","shell.execute_reply":"2022-08-10T10:41:03.244407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outlier_low = -1.5*(q3-q1)\noutlier_high= 1.5*(q3-q1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.246590Z","iopub.execute_input":"2022-08-10T10:41:03.247870Z","iopub.status.idle":"2022-08-10T10:41:03.258090Z","shell.execute_reply.started":"2022-08-10T10:41:03.247832Z","shell.execute_reply":"2022-08-10T10:41:03.256569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df[(df[\"Fare\"] > outlier_low ) & (df[\"Fare\"]<outlier_high)]\ndf.shape\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.259649Z","iopub.execute_input":"2022-08-10T10:41:03.260027Z","iopub.status.idle":"2022-08-10T10:41:03.276628Z","shell.execute_reply.started":"2022-08-10T10:41:03.259995Z","shell.execute_reply":"2022-08-10T10:41:03.275754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one hot encoding\ndf.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.278010Z","iopub.execute_input":"2022-08-10T10:41:03.278635Z","iopub.status.idle":"2022-08-10T10:41:03.300949Z","shell.execute_reply.started":"2022-08-10T10:41:03.278591Z","shell.execute_reply":"2022-08-10T10:41:03.299421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transfomed cols\n\npd.get_dummies(data=df ,columns=[\"Pclass\",\"Sex\",\"Embarked\",\"Family_type\"],drop_first=True).sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.302645Z","iopub.execute_input":"2022-08-10T10:41:03.303049Z","iopub.status.idle":"2022-08-10T10:41:03.333523Z","shell.execute_reply.started":"2022-08-10T10:41:03.303010Z","shell.execute_reply":"2022-08-10T10:41:03.332648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =pd.get_dummies(data=df ,columns=[\"Pclass\",\"Sex\",\"Embarked\",\"Family_type\"],drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.335082Z","iopub.execute_input":"2022-08-10T10:41:03.335424Z","iopub.status.idle":"2022-08-10T10:41:03.349559Z","shell.execute_reply.started":"2022-08-10T10:41:03.335392Z","shell.execute_reply":"2022-08-10T10:41:03.348218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(18,8))\nsns.heatmap(df.corr(),cmap=\"summer\",annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:03.351840Z","iopub.execute_input":"2022-08-10T10:41:03.352544Z","iopub.status.idle":"2022-08-10T10:41:04.112344Z","shell.execute_reply.started":"2022-08-10T10:41:03.352504Z","shell.execute_reply":"2022-08-10T10:41:04.110948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.clustermap(df.corr(),cmap=\"coolwarm\",annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T10:41:04.114096Z","iopub.execute_input":"2022-08-10T10:41:04.114542Z","iopub.status.idle":"2022-08-10T10:41:05.375179Z","shell.execute_reply.started":"2022-08-10T10:41:04.114497Z","shell.execute_reply":"2022-08-10T10:41:05.373801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"table\"></a>\n<h1 style=\"background-color:yellow;font-family:newtimeroman;font-size:200%;text-align:center;border-radius:50px;color:black\">Canclusion </h1>\n\n1. Chance of female sevival is higher then male survival\n2. Travling in pclass  was deadist\n3. somehow poeople going to C was survived more\n4. people in the range of 20 to 40 is higher chance of not surviving\n5. surviving more chance who with smaller family then alone and large falmily \n","metadata":{}},{"cell_type":"markdown","source":"1. if you like this notebook plz give upvote\n2. spelling mistakes are in the notebook so please","metadata":{}}]}