{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\npd.set_option('display.max_columns', None)  # or 1000\npd.set_option('display.max_colwidth', None)\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-11-25T21:24:56.455906Z","iopub.execute_input":"2021-11-25T21:24:56.456262Z","iopub.status.idle":"2021-11-25T21:24:56.462495Z","shell.execute_reply.started":"2021-11-25T21:24:56.456227Z","shell.execute_reply":"2021-11-25T21:24:56.461642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#WHAT ARE THE INDICATORS OF ADOPTION SPEED? statistical testing\n#PREDICT ADOPTION SPEED FOR TEST SET logistic regression or neural network\n#read data\n#sentiment data is provided but has less rows than the main dataset (how to use?)\ntrain = pd.read_csv(\"../input/petfinder-adoption-prediction/train/train.csv\")\ntest = pd.read_csv(\"../input/petfinder-adoption-prediction/test/test.csv\")\nbreed_labels = pd.read_csv(\"../input/petfinder-adoption-prediction/breed_labels.csv\")\ncolor_labels = pd.read_csv(\"../input/petfinder-adoption-prediction/color_labels.csv\")\nstate_labels = pd.read_csv(\"../input/petfinder-adoption-prediction/state_labels.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.257391Z","iopub.execute_input":"2021-11-25T20:30:38.258245Z","iopub.status.idle":"2021-11-25T20:30:38.636942Z","shell.execute_reply.started":"2021-11-25T20:30:38.258204Z","shell.execute_reply":"2021-11-25T20:30:38.636039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head() \ntrain.shape ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.638228Z","iopub.execute_input":"2021-11-25T20:30:38.638447Z","iopub.status.idle":"2021-11-25T20:30:38.646443Z","shell.execute_reply.started":"2021-11-25T20:30:38.638406Z","shell.execute_reply":"2021-11-25T20:30:38.645891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head() \n#test set is missing adoption speed column (dependent variable) as the data comes from a competition so in order to evaluate model performance later on, split train set into new train1 and test1 set to have dependent values for both samples\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.648091Z","iopub.execute_input":"2021-11-25T20:30:38.648871Z","iopub.status.idle":"2021-11-25T20:30:38.682432Z","shell.execute_reply.started":"2021-11-25T20:30:38.648825Z","shell.execute_reply":"2021-11-25T20:30:38.681853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"breed_labels.head() #type 1 refers to dogs and type 2 to cats\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.6833Z","iopub.execute_input":"2021-11-25T20:30:38.684005Z","iopub.status.idle":"2021-11-25T20:30:38.693672Z","shell.execute_reply.started":"2021-11-25T20:30:38.683973Z","shell.execute_reply":"2021-11-25T20:30:38.692676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"color_labels.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.694882Z","iopub.execute_input":"2021-11-25T20:30:38.695132Z","iopub.status.idle":"2021-11-25T20:30:38.709585Z","shell.execute_reply.started":"2021-11-25T20:30:38.695104Z","shell.execute_reply":"2021-11-25T20:30:38.708854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"state_labels.head() #join breed, color, and state labels with main dataset on ID for visual clarity (seeing black instead of 1 for example)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.711103Z","iopub.execute_input":"2021-11-25T20:30:38.711321Z","iopub.status.idle":"2021-11-25T20:30:38.725335Z","shell.execute_reply.started":"2021-11-25T20:30:38.711297Z","shell.execute_reply":"2021-11-25T20:30:38.724769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Wrangling","metadata":{}},{"cell_type":"code","source":"train_copy = train.copy() #make a copy to avoid changing raw data\n\n#rename columns to be used as key for merging then merge with main dataset train_copy\n#rename columns for clarity\nstate_labels = state_labels.rename(columns={'StateID':'State'}) \ntrain_copy = train_copy.merge(state_labels,on=\"State\",how='left') #left join to ensure rows aren't lost, can inspect null values after merging\n\nbreed_labels = breed_labels.rename(columns={'BreedID':'Breed1'})\ntrain_copy = train_copy.merge(breed_labels, on = 'Breed1', how = 'left')\ntrain_copy = train_copy.rename(columns={'BreedName':\"Primary Breed\"})\nbreed_labels = breed_labels.rename(columns={'Breed1':'Breed2'})\ntrain_copy = train_copy.merge(breed_labels, on = 'Breed2', how = 'left')\ntrain_copy = train_copy.rename(columns={'BreedName':'Secondary Breed','Type_x':'Species'})\nbreed_labels = breed_labels.rename(columns={'Breed2':'Breed1'}) #this is to avoid having to run code from start\ntrain_copy= train_copy.drop([\"Breed1\",'Breed2','Type_y','Type'],axis = 1)\ncolor_labels = color_labels.rename(columns={'ColorID':'Color1'})\ntrain_copy = train_copy.merge(color_labels, on = 'Color1', how = 'left')\ncolor_labels = color_labels.rename(columns={'Color1':'Color2'})\ntrain_copy = train_copy.merge(color_labels, on = 'Color2', how = 'left')\ncolor_labels = color_labels.rename(columns={'Color2':'Color3'})\ntrain_copy = train_copy.merge(color_labels, on = 'Color3', how = 'left')\ncolor_labels = color_labels.rename(columns={'Color3':'Color1'})\ntrain_copy = train_copy.rename(columns={'ColorName_x':'Color 1','ColorName_y':'Color 2','ColorName':'Color 3'})\ntrain_copy = train_copy.drop([\"Color1\",'Color2','Color3','State'],axis = 1)\ntrain_copy = train_copy.rename(columns={'Age':'Age (months)'})\nprint(train_copy.shape) #sanity check that dataset size is unchanged\n\n#rename categorical values for clarity\ntrain_copy[\"Gender\"] = train_copy[\"Gender\"].replace({1:'Male',2:'Female',3:'Mixed'}) #mixed refers to profiles with more than 1 pet\ntrain_copy[\"MaturitySize\"] = train_copy[\"MaturitySize\"].replace({1:'Small',2:'Medium',3:'Large',4:'Extra Large',0:'Not Specified'})\ntrain_copy[\"FurLength\"] = train_copy[\"FurLength\"].replace({1:'Short',2:'Medium',3:'Long',0:'Not Specified'})\ntrain_copy[\"Vaccinated\"] = train_copy[\"Vaccinated\"].replace({1:'Yes',2:'No',3:'Not Sure'})\ntrain_copy[\"Dewormed\"] = train_copy[\"Dewormed\"].replace({1:'Yes',2:'No',3:'Not Sure'})\ntrain_copy[\"Sterilized\"] = train_copy[\"Sterilized\"].replace({1:'Yes',2:'No',3:'Not Sure'})\ntrain_copy[\"Health\"] = train_copy[\"Health\"].replace({1:'Healthy',2:'Minor Injury',3:'Serious Injury',0:'Not Specified'})\ntrain_copy[\"Species\"] = train_copy[\"Species\"].replace({1:'Dog',2:'Cat'})\ntrain_copy[\"AdoptionSpeed\"]=train_copy[\"AdoptionSpeed\"].replace({0:'Adopted on the same day',1:'Adopted between 1-7 days',2:'Adopted between 8-30 days',3:'Adopted between 31-90 days',4:'No adoption after 90 days'})\n#treat no adoption after 90 days as no adoption since we have no data on whether they eventually get adopted or not\nprint(train_copy.columns)\ntrain_copy.head(25)\n\n#Adoption fee 0 not renamed to be free so that the values stay as integers\n\n#check each variable for transformations, null values, outliers\n#are there new variables of relevance that can be created?\n#plot those new variables if created\n#basic summary statistics for each variable\n#univariate plots \n\n#multivariate plots against target variable (adoption status and possibly adoption speed to get a sense of weights)\n#correlation map\n\ntrain_copy[['Quantity','Gender','Name']].head(25)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T21:05:45.861888Z","iopub.execute_input":"2021-11-25T21:05:45.862723Z","iopub.status.idle":"2021-11-25T21:05:46.029578Z","shell.execute_reply.started":"2021-11-25T21:05:45.86267Z","shell.execute_reply":"2021-11-25T21:05:46.028728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Cleaning and Initial Thoughts\nchecking for missing values, outliers, and unexpected values","metadata":{}},{"cell_type":"code","source":"plt.hist(train_copy[\"Species\"], color='blue',edgecolor='black', bins=2)\n#all values are either cat or dog or null\n#More Dogs are listed\nprint(train_copy['Species'].isnull().sum())\nprint(train_copy['Species'].describe())\n#no missing values reported and count (14993) matches with original dataset\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:38.917418Z","iopub.execute_input":"2021-11-25T20:30:38.918161Z","iopub.status.idle":"2021-11-25T20:30:39.146269Z","shell.execute_reply.started":"2021-11-25T20:30:38.918127Z","shell.execute_reply":"2021-11-25T20:30:39.145447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#want to drop name column\n#no name values are reported in several different ways (No name yet, NAN, lost dog etc.) making it difficult to remove\n#check with correlation map, but intuition says name shouldn't have much of an effect on adoption speed\n#want to drop rows with quantity > 1 (group profiles) as I am unsure how adoption speed is determined for group profiles (ex. is adoption speed determined once all pets in a group are adopted or after the first one gets adopted?)\nplt.hist(train_copy['Quantity'],color='blue', edgecolor='black', bins=[1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20])\n#single profiles still \ndisplay(train_copy[train_copy[\"Quantity\"] == 1])\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T21:25:06.239036Z","iopub.execute_input":"2021-11-25T21:25:06.239445Z","iopub.status.idle":"2021-11-25T21:25:20.672013Z","shell.execute_reply.started":"2021-11-25T21:25:06.239407Z","shell.execute_reply":"2021-11-25T21:25:20.671269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_copy['Age (months)'].isnull().sum())\nprint(train_copy['Age (months)'].describe())\n#all values in age column are integers, no missing values are reported, and count matches up (14993)\nprint(train_copy.boxplot('Age (months)', grid = False))\n#boxplot over histogram to more clearly see potential outliers\n#some extreme points around 250 months but that is still within a possible age range for pets (20 years)\n#median age is less than 15 months so there are much more young pets than there are old \n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:39.154841Z","iopub.execute_input":"2021-11-25T20:30:39.155414Z","iopub.status.idle":"2021-11-25T20:30:39.385714Z","shell.execute_reply.started":"2021-11-25T20:30:39.15537Z","shell.execute_reply":"2021-11-25T20:30:39.384753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_copy[\"Gender\"], color='blue',edgecolor='black')\nprint(train_copy['Gender'].isnull().sum())\nprint(train_copy['Gender'].describe())\n#no missing values reported\n#all values are male, female, or mixed \n#seems to be more female pets listed but could be similar count when including the mixed group","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:39.387257Z","iopub.execute_input":"2021-11-25T20:30:39.387558Z","iopub.status.idle":"2021-11-25T20:30:39.603433Z","shell.execute_reply.started":"2021-11-25T20:30:39.387511Z","shell.execute_reply":"2021-11-25T20:30:39.602433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_copy[\"MaturitySize\"], color='blue',edgecolor='black')\nprint(train_copy['MaturitySize'].isnull().sum())\nprint(train_copy['MaturitySize'].describe())\n#no missing or unexpected values\n#much more medium sized pets than there are for other sizes\n#break down by species and gender as well for the categorical variables\n#add detail to graph (titles etc.)\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:39.605633Z","iopub.execute_input":"2021-11-25T20:30:39.60644Z","iopub.status.idle":"2021-11-25T20:30:39.822956Z","shell.execute_reply.started":"2021-11-25T20:30:39.606402Z","shell.execute_reply":"2021-11-25T20:30:39.822225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_copy['FurLength'], color='blue',edgecolor='black')\nprint(train_copy['FurLength'].isnull().sum())\nprint(train_copy['FurLength'].describe())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:39.824378Z","iopub.execute_input":"2021-11-25T20:30:39.826885Z","iopub.status.idle":"2021-11-25T20:30:40.02638Z","shell.execute_reply.started":"2021-11-25T20:30:39.826846Z","shell.execute_reply":"2021-11-25T20:30:40.02554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.hist(train_copy['Vaccinated'], color='blue',edgecolor='black')\nprint(train_copy['Vaccinated'].isnull().sum())\nprint(train_copy['Vaccinated'].describe())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:40.027433Z","iopub.execute_input":"2021-11-25T20:30:40.027638Z","iopub.status.idle":"2021-11-25T20:30:40.369231Z","shell.execute_reply.started":"2021-11-25T20:30:40.027614Z","shell.execute_reply":"2021-11-25T20:30:40.368216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.hist(train_copy['Dewormed'], color='blue',edgecolor='black')\nprint(train_copy['Dewormed'].isnull().sum())\nprint(train_copy['Dewormed'].describe())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:40.370648Z","iopub.execute_input":"2021-11-25T20:30:40.371432Z","iopub.status.idle":"2021-11-25T20:30:40.601776Z","shell.execute_reply.started":"2021-11-25T20:30:40.371388Z","shell.execute_reply":"2021-11-25T20:30:40.600926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.hist(train_copy['Sterilized'], color='blue',edgecolor='black')\nprint(train_copy['Sterilized'].isnull().sum())\nprint(train_copy['Sterilized'].describe())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:40.602842Z","iopub.execute_input":"2021-11-25T20:30:40.603044Z","iopub.status.idle":"2021-11-25T20:30:40.809678Z","shell.execute_reply.started":"2021-11-25T20:30:40.60302Z","shell.execute_reply":"2021-11-25T20:30:40.808797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nplt.hist(train_copy['Health'], color='blue',edgecolor='black')\nprint(train_copy['Health'].isnull().sum())\nprint(train_copy['Health'].describe())","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:40.810796Z","iopub.execute_input":"2021-11-25T20:30:40.811015Z","iopub.status.idle":"2021-11-25T20:30:41.041923Z","shell.execute_reply.started":"2021-11-25T20:30:40.81099Z","shell.execute_reply":"2021-11-25T20:30:41.041086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(train_copy['Quantity'], color='blue',edgecolor='black')\nprint(train_copy['Quantity'].isnull().sum())\nprint(train_copy['Quantity'].describe())\ntrain_copy.boxplot('Quantity', grid = False)\n#hist shows distribution but hard to see outliers without smth like boxplot\n#want to drop profiles with more than 1 pet as their adoption speed might be misleading (is adoption speed determined by when the last pet in the group is adopted?)\n#might not need boxplot after doing multivariate scatter plots \n#figure out how to show figures side by side","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.043377Z","iopub.execute_input":"2021-11-25T20:30:41.044247Z","iopub.status.idle":"2021-11-25T20:30:41.288104Z","shell.execute_reply.started":"2021-11-25T20:30:41.044205Z","shell.execute_reply":"2021-11-25T20:30:41.287546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'Fee', ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.288984Z","iopub.execute_input":"2021-11-25T20:30:41.289469Z","iopub.status.idle":"2021-11-25T20:30:41.29463Z","shell.execute_reply.started":"2021-11-25T20:30:41.289438Z","shell.execute_reply":"2021-11-25T20:30:41.293888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'RescuerID', ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.295853Z","iopub.execute_input":"2021-11-25T20:30:41.296527Z","iopub.status.idle":"2021-11-25T20:30:41.307586Z","shell.execute_reply.started":"2021-11-25T20:30:41.296489Z","shell.execute_reply":"2021-11-25T20:30:41.306725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'VideoAmt', ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.30893Z","iopub.execute_input":"2021-11-25T20:30:41.309414Z","iopub.status.idle":"2021-11-25T20:30:41.318409Z","shell.execute_reply.started":"2021-11-25T20:30:41.309371Z","shell.execute_reply":"2021-11-25T20:30:41.317556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'Description', ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.319782Z","iopub.execute_input":"2021-11-25T20:30:41.320252Z","iopub.status.idle":"2021-11-25T20:30:41.331504Z","shell.execute_reply.started":"2021-11-25T20:30:41.320209Z","shell.execute_reply":"2021-11-25T20:30:41.330905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'PetID',\n       ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.332506Z","iopub.execute_input":"2021-11-25T20:30:41.333054Z","iopub.status.idle":"2021-11-25T20:30:41.343354Z","shell.execute_reply.started":"2021-11-25T20:30:41.333018Z","shell.execute_reply":"2021-11-25T20:30:41.342788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'PhotoAmt', ","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.344733Z","iopub.execute_input":"2021-11-25T20:30:41.345039Z","iopub.status.idle":"2021-11-25T20:30:41.354768Z","shell.execute_reply.started":"2021-11-25T20:30:41.344999Z","shell.execute_reply":"2021-11-25T20:30:41.354184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'AdoptionSpeed'","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.355788Z","iopub.execute_input":"2021-11-25T20:30:41.356102Z","iopub.status.idle":"2021-11-25T20:30:41.367806Z","shell.execute_reply.started":"2021-11-25T20:30:41.356066Z","shell.execute_reply":"2021-11-25T20:30:41.36707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#a number of variables have imbalanced classes (ex. healthy vs injured pets), one way to evaluate this issue could be to use precision, recall, and f1 scores as metrics over accuracy\n#standardization?\n#split after completing cleaning and transformations\n#random_seed = 2  \n#train0, val1 = train_test_split(train_copy, test_size = 0.2, random_state=random_seed) #split into main and validation dataset \n#train1, test1 = train_test_split(train0, test_size = 0.3, random_state=random_seed) #split main dataset into train1 and test1\n#print(train1.shape) #check for proper distribution\n#print(test1.shape)\n#print(val1.shape)\n#print(train0.shape)\n#train1.head()\n","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.368961Z","iopub.execute_input":"2021-11-25T20:30:41.369296Z","iopub.status.idle":"2021-11-25T20:30:41.378772Z","shell.execute_reply.started":"2021-11-25T20:30:41.369269Z","shell.execute_reply":"2021-11-25T20:30:41.377836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merge all datasets \n#check for regression assumptions (linearity, normality etc.)\n#standardization?\n#model linear regression vs neural network\n#for model choose between using keras or sklearn, lean towards keras for more thorough model building (less default hyperparameters)","metadata":{"execution":{"iopub.status.busy":"2021-11-25T20:30:41.382727Z","iopub.execute_input":"2021-11-25T20:30:41.38313Z","iopub.status.idle":"2021-11-25T20:30:41.388338Z","shell.execute_reply.started":"2021-11-25T20:30:41.383094Z","shell.execute_reply":"2021-11-25T20:30:41.387555Z"},"trusted":true},"execution_count":null,"outputs":[]}]}