{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nbase_dataset=pd.read_csv(\"../input/application_train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44e37040f6def2ee5a6bc557fa6051cfe9738c2d"},"cell_type":"code","source":"base_dataset.head()\nbase_dataset=base_dataset.sample(1000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ec3f32f3f486e8cfe149df5bd6c28c5f133a503"},"cell_type":"code","source":"base_dataset.reset_index(inplace=True)\nbase_dataset.drop(['index','SK_ID_CURR'],axis=1,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20f68212578c14fb178bf2ca13f2f912ac59852e"},"cell_type":"code","source":"base_dataset.shape\nbase_dataset.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9a51ecd905a56c6693bc0c66465e64795b42cab7"},"cell_type":"code","source":"base_dataset.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2296fddbeeefb2704d42b9a546f0594b57177a4c"},"cell_type":"code","source":"def nullvalue_function(base_dataset,percentage):\n    \n    # Checking the null value occurance\n    \n    print(base_dataset.isna().sum())\n\n    # Printing the shape of the data \n    \n    print(base_dataset.shape)\n    \n    # Converting  into percentage table\n    \n    null_value_table=pd.DataFrame((base_dataset.isna().sum()/base_dataset.shape[0])*100).sort_values(0,ascending=False )\n    \n    null_value_table.columns=['null percentage']\n    \n    # Defining the threashold values \n    \n    null_value_table[null_value_table['null percentage']>percentage].index\n    \n    # Drop the columns that has null values more than threashold \n    base_dataset.drop(null_value_table[null_value_table['null percentage']>30].index,axis=1,inplace=True)\n    \n    # Replace the null values with median() # continous variables \n    for i in base_dataset.describe().columns:\n        base_dataset[i].fillna(base_dataset[i].median(),inplace=True)\n    # Replace the null values with mode() #categorical variables\n    for i in base_dataset.describe(include='object').columns:\n        base_dataset[i].fillna(base_dataset[i].value_counts().index[0],inplace=True)\n  \n    print(base_dataset.shape)\n    \n    return base_dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"75f5e0589542c92d90f9be8d5386a1409db16cc9"},"cell_type":"code","source":"base_dataset_null=nullvalue_function(base_dataset,30)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d2285b4a6f01ce039968441bbdbdcf72eb23040"},"cell_type":"code","source":"base_dataset_null.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e98b5c47fbc552ef292c2ae6838292985a00f95"},"cell_type":"code","source":"base_dataset_null","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b39997312325ea1eef9fd7e1dd30d81d588bef7c"},"cell_type":"code","source":"from sklearn import preprocessing\n\ndef variables_creation(base_dataset,unique):\n    \n    cat=base_dataset.describe(include='object').columns\n    \n    cont=base_dataset.describe().columns\n    \n    x=[]\n    \n    for i in base_dataset[cat].columns:\n        if len(base_dataset[i].value_counts().index)<unique:\n            x.append(i)\n    \n    dummies_table=pd.get_dummies(base_dataset[x])\n    encode_table=base_dataset[x]\n    \n    le = preprocessing.LabelEncoder()\n    lable_encode=[]\n    \n    for i in encode_table.columns:\n        le.fit(encode_table[i])\n        le.classes_\n        lable_encode.append(le.transform(encode_table[i]))\n        \n    lable_encode=np.array(lable_encode)\n    lable=lable_encode.reshape(base_dataset.shape[0],len(x))\n    lable=pd.DataFrame(lable)\n    return (lable,dummies_table,cat,cont)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8bad9f07deba502e1f6f41e7e91a8262650b5398"},"cell_type":"code","source":"import numpy as np\n(lable,dummies_table,cat,cont)=variables_creation(base_dataset_null,8)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"062b64133ee1dbf7d8c7d07e64ddff1f66a6135a"},"cell_type":"code","source":"cat","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"300a0f5463a9551706e8f41321ed39669695aed0"},"cell_type":"code","source":"lable.rename(columns={0:'NAME_CONTRACT_TYPE'},inplace=True)\nlable.rename(columns={1:'CODE_GENDER'},inplace=True)\nlable.rename(columns={2:'FLAG_OWN_CAR'},inplace=True)\nlable.rename(columns={3:'FLAG_OWN_REALTY'},inplace=True)\nlable.rename(columns={4:'NAME_TYPE_SUITE'},inplace=True)\nlable.rename(columns={5:'NAME_INCOME_TYPE'},inplace=True)\nlable.rename(columns={6:'NAME_EDUCATION_TYPE'},inplace=True)\nlable.rename(columns={7:'NAME_FAMILY_STATUS'},inplace=True)\nlable.rename(columns={8:'NAME_HOUSING_TYPE'},inplace=True)\nlable.rename(columns={9:'WEEKDAY_APPR_PROCESS_START'},inplace=True)\nlable.rename(columns={10:'ORGANIZATION_TYPE'},inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68c99cf4f30d3bfc231a775a4a94a6c7de19b980"},"cell_type":"code","source":"for i in lable.columns:\n    base_dataset_null[i]=lable[i]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"30e86e8f0f11645aaff3c831c44ae1eb684bef1e"},"cell_type":"code","source":"base_dataset_null.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"50fe029b5fe9e9d77e21c640679b20cdf94232de"},"cell_type":"code","source":"base_dataset_null.drop('AMT_CREDIT',axis=1).columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f4085fa19028a414a0ad115effb00011e054f236"},"cell_type":"code","source":"base_dataset_null.var().sort_values(ascending=False).head(20).index[1:]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"632a3b7459d2cdc4220cf71b2a9553efe33a10e2"},"cell_type":"code","source":"def outliers(df):\n    import numpy as np\n    import statistics as sts\n\n    for i in df.describe().columns:\n        x=np.array(df[i])\n        p=[]\n        Q1 = df[i].quantile(0.25)\n        Q3 = df[i].quantile(0.75)\n        IQR = Q3 - Q1\n        LTV= Q1 - (1.5 * IQR)\n        UTV= Q3 + (1.5 * IQR)\n        for j in x:\n            if j <= LTV or j>=UTV:\n                p.append(sts.median(x))\n            else:\n                p.append(j)\n        df[i]=p\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ccf99a923859c11407c139113f45c3e1fa57f55"},"cell_type":"code","source":"base_dataset_null.shape\noutliers_treated=outliers(base_dataset_null[base_dataset_null.drop('AMT_CREDIT',axis=1).columns])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e33d83e6d26cbf6ff393692b8a476ae1af477730"},"cell_type":"code","source":"outliers_treated.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a2d728183005ae071def31c19d59c6994c9d098"},"cell_type":"code","source":"def univariate_analysis(base_null_value_treated):\n    import matplotlib.pyplot as plt\n    col=[]\n    for i in base_null_value_treated.describe().columns:\n        var=base_null_value_treated[i].value_counts().values.var()\n        col.append([i,var])\n        variance_table=pd.DataFrame(col)\n        variance_table[variance_table[1]>100][0].values\n    return variance_table[variance_table[1]>100][0].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f0d6b840135e2d289b3a9558ee4e1ccd0e6d152"},"cell_type":"code","source":"viz=outliers_treated[['AMT_INCOME_TOTAL','AMT_ANNUITY','AMT_GOODS_PRICE']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7a0f5556c9a5ae8a1d3a5d6042e7efd85ba0e50"},"cell_type":"code","source":"df_columns=univariate_analysis(viz)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"797f6571ac1ea6729bd07cde6c94787ac4aa51da"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nfor i in df_columns:\n    plt.hist(outliers_treated[i])\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a3eb2f4546c00c46e3a007dc72ae21629af92345"},"cell_type":"code","source":"outliers_treated.columns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fba6dadd7d34779d5be70526421c0b9b34ed8847"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfor i in df_columns:\n    for j in df_columns:\n        if i!=j:\n            sns.jointplot(outliers_treated[i],outliers_treated[j])\n            plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d507dbf488f86a856378ee14faf2fbe818c7ca9"},"cell_type":"code","source":"outliers_treated=outliers_treated[outliers_treated.describe().columns]\noutliers_treated['const']=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9f574c2d9dbfa8f03ca517e72d24c941dfe5e8d"},"cell_type":"code","source":"outliers_treated['target']=base_dataset_null['AMT_CREDIT']\ny=outliers_treated['target']\nx=outliers_treated.drop('target',axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a1647e02d72423bd5cf387c1afdc02dfb32d2d7"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(x, y, test_size=0.20, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5e7a12ac878b0c594d3c15c99235888b8cec0993"},"cell_type":"code","source":"print(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e90650d4544fde1429667664afd6f8f3492704f3"},"cell_type":"code","source":"\nfrom sklearn.linear_model import LinearRegression\nlm=LinearRegression()\nlm.fit(X_train,y_train)\nlm.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ade04cf39df3925b4dca5a1061c86612893071a6"},"cell_type":"code","source":"predicted_values=lm.predict(X_test)\nsum(abs(predicted_values-y_test.values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"25f6314e91591d3234a7ce11a46f413d8fb8bc87"},"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\nMAE=mean_absolute_error(y_test.values,predicted_values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1230cb79ad03c3aee5e20f63b49f15d222ba08c2"},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nMSE=mean_squared_error(y_test.values,predicted_values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6d077f7e6c213ff3c97673de452378d6af4fd5dc"},"cell_type":"code","source":"from sklearn.metrics import mean_squared_error\nRMSE=np.sqrt(mean_squared_error(y_test.values,predicted_values))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"738da5318c477d49f6745ed2749b886a1f039954"},"cell_type":"code","source":"MAPE=sum(abs((y_test.values-predicted_values)/(y_test.values)))/X_test.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aeead5cde25d419a1c7d1623ff186c7dfa886b62"},"cell_type":"code","source":"def regression_model(predicted_values,y_test):\n    from sklearn.metrics import mean_absolute_error\n    from sklearn.metrics import mean_squared_error\n    from sklearn.metrics import r2_score\n    total_error=sum(abs(predicted_values-y_test.values))\n    MSE=mean_absolute_error(y_test.values,predicted_values)\n    MAE=mean_squared_error(y_test.values,predicted_values)\n    RMSE=np.sqrt(mean_squared_error(y_test.values,predicted_values))\n    MAPE=sum(abs((y_test.values-predicted_values)/(y_test.values)))/X_test.shape[0]\n    r2=r2_score(predicted_values,y_test)\n    print(\"total error\",total_error)\n    print(\"MSE\",MSE)\n    print(\"MAE\",MAE)\n    print(\"RMSE\",RMSE)\n    print(\"MAPE\",MAPE)\n    print(\"R2\",r2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8367ffa6c062885529c69ddf203c666b1387bdc"},"cell_type":"code","source":"regression_model(predicted_values,y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"17307deba0b96a08f48107a64dcddd60d5971add"},"cell_type":"code","source":"error_table=pd.DataFrame(lm.predict(X_test),y_test.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c98c5a631b43d8c8ab2a195ae500b1133c47d9de"},"cell_type":"code","source":"error_table.reset_index(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2ee8ee4ceb91e3862b3a911419aece920373d08c"},"cell_type":"code","source":"error_table.columns=['pred','actual']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d21db5c2cfdef0842a753d6d9607058bc454b85"},"cell_type":"code","source":"error_table.plot(figsize=(20,8))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a367174175d850e26252c8164012fe06b13da3fb"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}