{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-14T11:21:35.597737Z","iopub.execute_input":"2022-06-14T11:21:35.598214Z","iopub.status.idle":"2022-06-14T11:21:36.36319Z","shell.execute_reply.started":"2022-06-14T11:21:35.598127Z","shell.execute_reply":"2022-06-14T11:21:36.362068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_obj(filepath):\n    with open(filepath, 'rb') as file:\n        obj = pickle.load(file)\n    return obj","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:36.366506Z","iopub.execute_input":"2022-06-14T11:21:36.367394Z","iopub.status.idle":"2022-06-14T11:21:36.373466Z","shell.execute_reply.started":"2022-06-14T11:21:36.367342Z","shell.execute_reply":"2022-06-14T11:21:36.372684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_encoder = load_obj(\"../input/amex-datasetcategorical-encoders/cat_encoder.pkl\")\ncustomer2id = load_obj(\"../input/amex-datasetcategorical-encoders/customer2id.pkl\")\nid2customer = load_obj(\"../input/amex-datasetcategorical-encoders/id2customer.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:36.375094Z","iopub.execute_input":"2022-06-14T11:21:36.375463Z","iopub.status.idle":"2022-06-14T11:21:37.676748Z","shell.execute_reply.started":"2022-06-14T11:21:36.375432Z","shell.execute_reply":"2022-06-14T11:21:37.675549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"count_df = pd.read_pickle(\"../input/amex-eda-data/all_counts_df.pkl\")\nna_df = pd.read_pickle(\"../input/amex-eda-data/all_na_df.pkl\")\ntrain_label = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\n\nprint(\"max number of observation per customer:\", count_df.num_records.max())","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:37.679031Z","iopub.execute_input":"2022-06-14T11:21:37.679376Z","iopub.status.idle":"2022-06-14T11:21:42.741907Z","shell.execute_reply.started":"2022-06-14T11:21:37.679346Z","shell.execute_reply":"2022-06-14T11:21:42.740442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label['customer_ID'] = train_label['customer_ID'].apply(lambda v: customer2id[v])\ntrain_label.set_index('customer_ID', inplace=True)\nna_df = na_df.merge(train_label, on='customer_ID')\n\nna_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:42.743674Z","iopub.execute_input":"2022-06-14T11:21:42.744149Z","iopub.status.idle":"2022-06-14T11:21:43.930137Z","shell.execute_reply.started":"2022-06-14T11:21:42.744105Z","shell.execute_reply":"2022-06-14T11:21:43.928518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"na_stat_df = []\nnum_customers = len(na_df)\n\nfor colname in na_df.columns:\n    percent_na = na_df[colname].sum()/num_customers/13\n    na_stat_df.append({\n        'colname': colname,\n        \"percent_na\": percent_na\n    })\n\n    \nna_stat_df = pd.DataFrame.from_dict(na_stat_df)\nna_stat_df = na_stat_df.sort_values('percent_na', ascending=False)\n\nna_stat_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:43.93184Z","iopub.execute_input":"2022-06-14T11:21:43.932252Z","iopub.status.idle":"2022-06-14T11:21:44.090131Z","shell.execute_reply.started":"2022-06-14T11:21:43.932217Z","shell.execute_reply":"2022-06-14T11:21:44.088825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of columns with 30% missing values:\", len(na_stat_df[na_stat_df.percent_na>0.3]))\nprint()\nna_stat_df.percent_na.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:44.091766Z","iopub.execute_input":"2022-06-14T11:21:44.092125Z","iopub.status.idle":"2022-06-14T11:21:44.109684Z","shell.execute_reply.started":"2022-06-14T11:21:44.092094Z","shell.execute_reply":"2022-06-14T11:21:44.108408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"distribution of the percent of missing values in the features.\")\nsns.histplot(na_stat_df['percent_na'], bins=10)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:44.111901Z","iopub.execute_input":"2022-06-14T11:21:44.112353Z","iopub.status.idle":"2022-06-14T11:21:44.356646Z","shell.execute_reply.started":"2022-06-14T11:21:44.112319Z","shell.execute_reply":"2022-06-14T11:21:44.355445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 7))\nax = sns.barplot(data=na_stat_df.head(35), x='colname', y='percent_na' )\n\nplt.yticks(np.arange(0, 1.0, 0.1))\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:44.357923Z","iopub.execute_input":"2022-06-14T11:21:44.358235Z","iopub.status.idle":"2022-06-14T11:21:44.751701Z","shell.execute_reply.started":"2022-06-14T11:21:44.358207Z","shell.execute_reply":"2022-06-14T11:21:44.750288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets pick the above top30 features and check their importance towards the target.","metadata":{}},{"cell_type":"code","source":"missing_columns = list(na_stat_df[na_stat_df.percent_na>0.3].colname.values)\ndata=[]\nfor colname in missing_columns:\n    df = na_df[na_df[colname] > 0]\n    den_ = df[colname].sum()\n    \n    df = (df.groupby('target')[[colname]].sum()/den_).reset_index()\n    \n    for _,row in df.iterrows():\n        target = row.target\n        v = row[colname]\n        \n        data.append({\n            'colname': colname,\n            'target': target,\n            'v': v\n        })","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:44.756006Z","iopub.execute_input":"2022-06-14T11:21:44.756349Z","iopub.status.idle":"2022-06-14T11:21:53.076027Z","shell.execute_reply.started":"2022-06-14T11:21:44.756322Z","shell.execute_reply":"2022-06-14T11:21:53.075278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_stat_df = pd.DataFrame.from_dict(data)\nmissing_stat_df = pd.pivot(data=missing_stat_df, index='colname', columns='target', values='v')\nmissing_stat_df.columns = ['target_0', 'target_1' ]\nmissing_stat_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:53.077031Z","iopub.execute_input":"2022-06-14T11:21:53.077854Z","iopub.status.idle":"2022-06-14T11:21:53.096741Z","shell.execute_reply.started":"2022-06-14T11:21:53.077777Z","shell.execute_reply":"2022-06-14T11:21:53.095565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_stat_df.merge(na_stat_df, on='colname').sort_values('percent_na')","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:53.099526Z","iopub.execute_input":"2022-06-14T11:21:53.100415Z","iopub.status.idle":"2022-06-14T11:21:53.126946Z","shell.execute_reply.started":"2022-06-14T11:21:53.10036Z","shell.execute_reply":"2022-06-14T11:21:53.125917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. it looks like we can discard a few features with high percent of miissing values.\n2. the train target ratio is (75/25) split of target(0/1).\n3. for feature: D_77 if we pick all na values--> 66% belongs to target 0, 34% to target1.\n   that looks to be significantly different from the overall distribution of (75, 25).","metadata":{}},{"cell_type":"markdown","source":"# Feature columns","metadata":{}},{"cell_type":"code","source":"%%time\nfeature_df = pd.read_pickle(\"../input/amex-eda-data/all_features_df.pkl\")\nfeature_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:21:53.128337Z","iopub.execute_input":"2022-06-14T11:21:53.128667Z","iopub.status.idle":"2022-06-14T11:22:05.744243Z","shell.execute_reply.started":"2022-06-14T11:21:53.128636Z","shell.execute_reply":"2022-06-14T11:22:05.74305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remaining_feat_columns = list(set(na_df.columns) - set(missing_columns) - set(['target']))\nremaining_feat_columns = [colname+\"_mean\" for colname in remaining_feat_columns]\n\nprint(\"number of remaning features:\", len(remaining_feat_columns))","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:05.745674Z","iopub.execute_input":"2022-06-14T11:22:05.746Z","iopub.status.idle":"2022-06-14T11:22:05.753412Z","shell.execute_reply.started":"2022-06-14T11:22:05.745971Z","shell.execute_reply":"2022-06-14T11:22:05.752031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data=[]\n\nfor colname in remaining_feat_columns:\n    s = feature_df[colname]\n    s = s[s.isna()==False]\n    data.append({\n        'feat_name': colname,\n        'avg': np.mean(s),\n        'std': np.std(s),\n        \n        'feat_min': np.min(s),\n        'feat_q01': np.quantile(s, 0.01),\n        'feat_q90': np.quantile(s, 0.9),\n        'feat_q99': np.quantile(s, 0.99),\n        'feat_max': np.max(s)\n    })\n\ndf = pd.DataFrame.from_dict(data)\ndf['r_max'] = df['feat_max'].div(df['feat_q99'])\ndf['r_min'] = np.abs(df['feat_min'].div(df['feat_q01']+1e-9))\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:05.754989Z","iopub.execute_input":"2022-06-14T11:22:05.755383Z","iopub.status.idle":"2022-06-14T11:22:09.51089Z","shell.execute_reply.started":"2022-06-14T11:22:05.755339Z","shell.execute_reply":"2022-06-14T11:22:09.509941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.avg.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:09.512077Z","iopub.execute_input":"2022-06-14T11:22:09.51286Z","iopub.status.idle":"2022-06-14T11:22:09.522532Z","shell.execute_reply.started":"2022-06-14T11:22:09.512827Z","shell.execute_reply":"2022-06-14T11:22:09.521787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['std'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:09.523875Z","iopub.execute_input":"2022-06-14T11:22:09.524936Z","iopub.status.idle":"2022-06-14T11:22:09.537576Z","shell.execute_reply.started":"2022-06-14T11:22:09.52489Z","shell.execute_reply":"2022-06-14T11:22:09.536895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"distribution of the mean of all the features.\")\nplt.hist(df.avg, bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:09.539118Z","iopub.execute_input":"2022-06-14T11:22:09.539928Z","iopub.status.idle":"2022-06-14T11:22:09.861907Z","shell.execute_reply.started":"2022-06-14T11:22:09.539897Z","shell.execute_reply":"2022-06-14T11:22:09.860448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"distribution of the std of all the features.\")\nplt.hist(df['std'], bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:09.863372Z","iopub.execute_input":"2022-06-14T11:22:09.863858Z","iopub.status.idle":"2022-06-14T11:22:10.16956Z","shell.execute_reply.started":"2022-06-14T11:22:09.86382Z","shell.execute_reply":"2022-06-14T11:22:10.168272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"feature relation between the rmax vs std\")\nplt.scatter(np.log(1+df['r_max']), df['std'] )\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:10.171094Z","iopub.execute_input":"2022-06-14T11:22:10.171549Z","iopub.status.idle":"2022-06-14T11:22:10.372949Z","shell.execute_reply.started":"2022-06-14T11:22:10.171513Z","shell.execute_reply":"2022-06-14T11:22:10.372012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Features with more variation are generally considered as some of the important features.\n2. But the above figure indicates the impact of choosing features with more variance under outliers.","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots(2, 2, figsize=(15, 5))\n\nsns.boxplot(data=df, x='feat_min', ax=ax[0, 0])\nsns.boxplot(data=df, x='feat_q01', ax=ax[0, 1])\nsns.boxplot(data=df, x='feat_max', ax=ax[1, 0])\nsns.boxplot(data=df, x='feat_q99', ax=ax[1, 1])\n\nax[0, 0].set_title(\"feat_min\")\nax[0, 1].set_title(\"feat_q01\")\nax[1, 0].set_title(\"feat_max\")\nax[1, 1].set_title(\"feat_q99\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:10.374309Z","iopub.execute_input":"2022-06-14T11:22:10.374662Z","iopub.status.idle":"2022-06-14T11:22:10.795297Z","shell.execute_reply.started":"2022-06-14T11:22:10.374632Z","shell.execute_reply":"2022-06-14T11:22:10.794562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There are a features which had the outliers in corresponding to their respecitve quantile[1%, 99%]","metadata":{}},{"cell_type":"code","source":"df.r_min.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:10.796638Z","iopub.execute_input":"2022-06-14T11:22:10.797228Z","iopub.status.idle":"2022-06-14T11:22:10.808962Z","shell.execute_reply.started":"2022-06-14T11:22:10.797186Z","shell.execute_reply":"2022-06-14T11:22:10.807715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.r_max.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:10.810467Z","iopub.execute_input":"2022-06-14T11:22:10.811027Z","iopub.status.idle":"2022-06-14T11:22:10.829082Z","shell.execute_reply.started":"2022-06-14T11:22:10.810977Z","shell.execute_reply":"2022-06-14T11:22:10.828116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\" feat (vs) log(r_max)\")\nplt.xlabel(\"feat no\")\nplt.ylabel(\"log (r_max) \")\nplt.yticks(np.arange(0, 10, 1.5))\nplt.plot(np.arange(len(df)), np.log(df.r_max.sort_values()))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:10.830842Z","iopub.execute_input":"2022-06-14T11:22:10.832068Z","iopub.status.idle":"2022-06-14T11:22:11.224506Z","shell.execute_reply.started":"2022-06-14T11:22:10.832014Z","shell.execute_reply":"2022-06-14T11:22:11.223697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\" feat (vs) log(r_min)\")\nplt.xlabel(\"feat no\")\nplt.ylabel(\"log (r_max) \")\nplt.yticks(np.arange(0, 10, 1.5))\nplt.plot(np.arange(len(df)), np.log(1+df.r_min.sort_values()))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:11.225953Z","iopub.execute_input":"2022-06-14T11:22:11.226492Z","iopub.status.idle":"2022-06-14T11:22:11.404866Z","shell.execute_reply.started":"2022-06-14T11:22:11.226458Z","shell.execute_reply":"2022-06-14T11:22:11.404216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"1. There are outliers in the values of features in both the maximum and minimum.\n2. Outliers:\n   any record for which the r_min and r_max >4x to the 99th,01st percentile\n3. we can clip the values of the outlliers to np.clip(x, min_threshhold, max_threshhold)","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:11.406172Z","iopub.execute_input":"2022-06-14T11:22:11.406655Z","iopub.status.idle":"2022-06-14T11:22:11.422801Z","shell.execute_reply.started":"2022-06-14T11:22:11.406625Z","shell.execute_reply":"2022-06-14T11:22:11.422011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets look at few plots before and after value clipping.","metadata":{}},{"cell_type":"code","source":"feat_threshold={}\nfor _,row in df.iterrows():\n    feat_name=row.feat_name\n    q01 = row.feat_q01\n    q99 = row.feat_q99\n    \n    feat_threshold[feat_name] = {}\n    feat_threshold[feat_name]['vmin'] = q01\n    feat_threshold[feat_name]['vmax'] = q99\n","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:11.424105Z","iopub.execute_input":"2022-06-14T11:22:11.424748Z","iopub.status.idle":"2022-06-14T11:22:11.448542Z","shell.execute_reply.started":"2022-06-14T11:22:11.424713Z","shell.execute_reply":"2022-06-14T11:22:11.447644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[(df.r_max > 100) | (df.r_min > 100)].shape","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:11.452835Z","iopub.execute_input":"2022-06-14T11:22:11.453638Z","iopub.status.idle":"2022-06-14T11:22:11.46564Z","shell.execute_reply.started":"2022-06-14T11:22:11.453581Z","shell.execute_reply":"2022-06-14T11:22:11.46452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_names = df[(df.r_max > 10) | (df.r_min > 10)].feat_name.values\nprint(\"number of features with outliers:\", len(feat_names))\n\nfor i, feat_name in enumerate(feat_names):\n    vmin = feat_threshold[feat_name]['vmin']\n    vmax = feat_threshold[feat_name]['vmax']\n    \n    _, ax = plt.subplots(1, 2, figsize=(12, 3))\n    ax[0].set_title(feat_name)\n    \n    ax[0].hist(feature_df[feat_name], bins=100)\n    ax[1].hist(np.clip(feature_df[feat_name], vmin, vmax ), bins=100)\n    \n    ax[0].set_xticks([])\n    ax[0].set_yticks([])\n    ax[1].set_xticks([])\n    ax[1].set_yticks([])\n    plt.show()\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:11.467447Z","iopub.execute_input":"2022-06-14T11:22:11.468088Z","iopub.status.idle":"2022-06-14T11:22:41.508094Z","shell.execute_reply.started":"2022-06-14T11:22:11.46805Z","shell.execute_reply":"2022-06-14T11:22:41.506994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lets clip the features and check the mean and variance values.","metadata":{}},{"cell_type":"code","source":"data=[]\n\nfor feat_name in remaining_feat_columns:\n    vmin = feat_threshold[feat_name]['vmin']\n    vmax = feat_threshold[feat_name]['vmax']\n    \n    s = feature_df[feat_name]\n    s = s[s.isna()==False]\n    s = np.clip(s, vmin, vmax)\n    \n    data.append({\n        'feat_name': feat_name,\n        'avg': np.mean(s),\n        'std': np.std(s),\n        \n        'feat_min': np.min(s),\n        'feat_q01': np.quantile(s, 0.01),\n        'feat_q90': np.quantile(s, 0.9),\n        'feat_q99': np.quantile(s, 0.99),\n        'feat_max': np.max(s)\n    })\n\ndf = pd.DataFrame.from_dict(data)\ndf['r_min'] = df['feat_min'].div(df['feat_q01'])\ndf['r_max'] = df['feat_max'].div(df['feat_q99'])\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:41.509671Z","iopub.execute_input":"2022-06-14T11:22:41.510685Z","iopub.status.idle":"2022-06-14T11:22:45.919774Z","shell.execute_reply.started":"2022-06-14T11:22:41.510633Z","shell.execute_reply":"2022-06-14T11:22:45.918638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.avg.describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:45.92101Z","iopub.execute_input":"2022-06-14T11:22:45.921365Z","iopub.status.idle":"2022-06-14T11:22:45.933488Z","shell.execute_reply.started":"2022-06-14T11:22:45.921325Z","shell.execute_reply":"2022-06-14T11:22:45.932277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['std'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:45.935129Z","iopub.execute_input":"2022-06-14T11:22:45.935563Z","iopub.status.idle":"2022-06-14T11:22:45.958193Z","shell.execute_reply.started":"2022-06-14T11:22:45.935526Z","shell.execute_reply":"2022-06-14T11:22:45.957232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"distribution of the mean of all the features.\")\nplt.hist(df.avg, bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:45.960723Z","iopub.execute_input":"2022-06-14T11:22:45.961194Z","iopub.status.idle":"2022-06-14T11:22:46.353836Z","shell.execute_reply.started":"2022-06-14T11:22:45.96115Z","shell.execute_reply":"2022-06-14T11:22:46.352749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"distribution of the std of all the features.\")\nplt.hist(df['std'], bins=100)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:46.355122Z","iopub.execute_input":"2022-06-14T11:22:46.356372Z","iopub.status.idle":"2022-06-14T11:22:46.708724Z","shell.execute_reply.started":"2022-06-14T11:22:46.35633Z","shell.execute_reply":"2022-06-14T11:22:46.707682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.sort_values('std')","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:46.709973Z","iopub.execute_input":"2022-06-14T11:22:46.711093Z","iopub.status.idle":"2022-06-14T11:22:46.733992Z","shell.execute_reply.started":"2022-06-14T11:22:46.711058Z","shell.execute_reply":"2022-06-14T11:22:46.733085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# lets plot 20 features with low variance and high variance","metadata":{}},{"cell_type":"code","source":"low_variance_features = df.sort_values('std').head(20).feat_name.values\nhigh_variance_features = df.sort_values('std').tail(20).feat_name.values","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:46.735582Z","iopub.execute_input":"2022-06-14T11:22:46.73624Z","iopub.status.idle":"2022-06-14T11:22:46.743452Z","shell.execute_reply.started":"2022-06-14T11:22:46.736196Z","shell.execute_reply":"2022-06-14T11:22:46.74273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, feat_name in enumerate(low_variance_features):\n    vmin = feat_threshold[feat_name]['vmin']\n    vmax = feat_threshold[feat_name]['vmax']\n    \n    _, ax = plt.subplots(1, 2, figsize=(12, 3))\n    ax[0].set_title(feat_name)\n    \n    ax[0].hist(feature_df[feat_name], bins=100)\n    ax[1].hist(np.clip(feature_df[feat_name], vmin, vmax ), bins=100)\n    \n    ax[0].set_xticks([])\n    ax[0].set_yticks([])\n    ax[1].set_yticks([])\n    \n    plt.show()\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:46.744703Z","iopub.execute_input":"2022-06-14T11:22:46.745147Z","iopub.status.idle":"2022-06-14T11:22:58.093036Z","shell.execute_reply.started":"2022-06-14T11:22:46.745107Z","shell.execute_reply":"2022-06-14T11:22:58.091849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, feat_name in enumerate(high_variance_features):\n    vmin = feat_threshold[feat_name]['vmin']\n    vmax = feat_threshold[feat_name]['vmax']\n    \n    _, ax = plt.subplots(1, 2, figsize=(12, 3))\n    ax[0].set_title(feat_name)\n    \n    ax[0].hist(feature_df[feat_name], bins=100)\n    ax[1].hist(np.clip(feature_df[feat_name], vmin, vmax ), bins=100)\n    \n    ax[0].set_xticks([])\n    ax[0].set_yticks([])\n    ax[1].set_yticks([])\n    \n    plt.show()\n    print()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:22:58.094986Z","iopub.execute_input":"2022-06-14T11:22:58.095445Z","iopub.status.idle":"2022-06-14T11:23:09.303628Z","shell.execute_reply.started":"2022-06-14T11:22:58.095402Z","shell.execute_reply":"2022-06-14T11:23:09.302662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_df[['B_31_mean','D_93_mean','R_24_mean']].describe()","metadata":{"execution":{"iopub.status.busy":"2022-06-14T11:23:09.305232Z","iopub.execute_input":"2022-06-14T11:23:09.306453Z","iopub.status.idle":"2022-06-14T11:23:10.648092Z","shell.execute_reply.started":"2022-06-14T11:23:09.3064Z","shell.execute_reply":"2022-06-14T11:23:10.646904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the low variance features\n1. B_31 : had many values as -1\n2. Many features had their values skewed to the extremes\n3. After cleaning a bit few features had normal kind of distribution","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. ","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}