{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport pickle\nimport gc\nimport seaborn as sns\n\nimport matplotlib.pyplot as plt\nfrom scipy.stats import kurtosis,skew, norm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-22T04:35:46.824494Z","iopub.execute_input":"2022-06-22T04:35:46.824909Z","iopub.status.idle":"2022-06-22T04:35:46.831114Z","shell.execute_reply.started":"2022-06-22T04:35:46.824874Z","shell.execute_reply":"2022-06-22T04:35:46.829857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"../input/amex-datasetcategorical-encoders/train_customer2id.pkl\", 'rb') as file:\n    train_customer2id = pickle.load(file)\nprint(len(train_customer2id))","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:35:46.851543Z","iopub.execute_input":"2022-06-22T04:35:46.852273Z","iopub.status.idle":"2022-06-22T04:35:47.615864Z","shell.execute_reply.started":"2022-06-22T04:35:46.852236Z","shell.execute_reply":"2022-06-22T04:35:47.614535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_df = pd.read_parquet(\"../input/amex-traindataset/train_dataset.parquet\")\n\ntrain_label = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\ntrain_label.customer_ID=train_label.customer_ID.apply(lambda k: train_customer2id[k])\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:35:47.617506Z","iopub.execute_input":"2022-06-22T04:35:47.617846Z","iopub.status.idle":"2022-06-22T04:36:23.838858Z","shell.execute_reply.started":"2022-06-22T04:35:47.617816Z","shell.execute_reply":"2022-06-22T04:36:23.837946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_uniform_qqplot(s, min_value, max_value):\n    y = []\n    for q in np.arange(0, 1.0, 0.05):\n        y.append(np.quantile(s, q) )\n\n    d = (max_value-min_value)\n    if d!=0:\n        x = np.arange(min_value, max_value, d/20)\n    else:\n        x =[min_value] * len(y)\n    \n    return (x, y)\n\ndef get_normal_qqplot(s):\n    mean = np.mean(s)\n    scale = np.std(s)\n    z = (s-mean)/scale\n    \n    x=[]\n    y=[]\n    for q in np.arange(0.01, 1.0, 0.05):\n        x.append(norm.ppf(q))\n        y.append(np.quantile(z, q))\n    return (x,y)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:36:23.848249Z","iopub.execute_input":"2022-06-22T04:36:23.848898Z","iopub.status.idle":"2022-06-22T04:36:23.860614Z","shell.execute_reply.started":"2022-06-22T04:36:23.848859Z","shell.execute_reply":"2022-06-22T04:36:23.859436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_quantile_plot(featname):\n    s = train_df[featname]\n    s = s[s.isna()==False]\n    \n    q01 = np.quantile(s, 0.01)\n    q99 = np.quantile(s, 0.99)\n    d = q99 - q01\n    s = np.clip(s, q01-2*d, q99+2*d)\n    \n    min_value = np.min(s)\n    max_value = np.max(s)\n    \n    x = [0]\n    y = [min_value]\n    for q in np.arange(0.05, 1.0, 0.05):\n        x.append(q)\n        y.append(np.quantile(s, q))\n    \n    x.append(1)\n    y.append(max_value)\n    \n    \n    fig, ax=plt.subplots(1, 3, figsize=(14, 6))\n    fig.suptitle(featname)\n    \n    ax[0].set_title(\"Quantile graph\")\n    ax[0].plot(x, y, marker=\"s\")\n    ax[0].set_xticks(np.arange(0, 1.0, 0.1))\n    \n    \n    (x, y) = get_uniform_qqplot(s, min_value, max_value)\n    ax[1].set_title(\"Q-Q graph for uniform distribution\")\n    ax[1].plot(x, y, label=\"q-q plot\", marker='s', color='blue')\n    ax[1].plot(x, x, label=\"line: 45\", color='green')\n    ax[1].legend(loc='best')\n    \n    (x, y) = get_normal_qqplot(s)\n    ax[2].set_title(\"Q-Q graph for normal distribution\")\n    ax[2].plot(x, y, label=\"q-q plot\", marker='s', color='blue')\n    ax[2].plot(x, x, label=\"line: 45\", color='green')\n    ax[2].legend(loc='best')\n    \n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:36:23.862802Z","iopub.execute_input":"2022-06-22T04:36:23.863632Z","iopub.status.idle":"2022-06-22T04:36:23.879448Z","shell.execute_reply.started":"2022-06-22T04:36:23.863579Z","shell.execute_reply":"2022-06-22T04:36:23.878288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features=['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', \n              'D_126', 'D_63',  'D_64', 'D_66', 'D_68'] + ['customer_ID', 'S_2', 'target']\nnumeric_features = [colname for colname in train_df.columns if colname not in cat_features ]","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:36:56.129938Z","iopub.execute_input":"2022-06-22T04:36:56.131067Z","iopub.status.idle":"2022-06-22T04:36:56.145848Z","shell.execute_reply.started":"2022-06-22T04:36:56.131005Z","shell.execute_reply":"2022-06-22T04:36:56.144628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for featname in sorted(numeric_features):\n    get_quantile_plot(featname)","metadata":{"execution":{"iopub.status.busy":"2022-06-22T04:37:44.951192Z","iopub.execute_input":"2022-06-22T04:37:44.952257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}