{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom lightgbm import LGBMRegressor\nimport optuna.integration.lightgbm as lgb\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport datetime\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n!pip install japanize_matplotlib\nimport japanize_matplotlib\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T11:47:09.530974Z","iopub.execute_input":"2022-07-09T11:47:09.532063Z","iopub.status.idle":"2022-07-09T11:47:24.601462Z","shell.execute_reply.started":"2022-07-09T11:47:09.531895Z","shell.execute_reply":"2022-07-09T11:47:24.60014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True, inplace=True):\n    if inplace:\n        reduce_mem_usage_inplace(df, verbose)\n        return\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2 \n    dfs = []\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    dfs.append(df[col].astype(np.int8))\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    dfs.append(df[col].astype(np.int16))\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    dfs.append(df[col].astype(np.int32))\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    dfs.append(df[col].astype(np.int64) ) \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    dfs.append(df[col].astype(np.float16))\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    dfs.append(df[col].astype(np.float32))\n                else:\n                    dfs.append(df[col].astype(np.float64))\n        else:\n            dfs.append(df[col])\n    \n    df_out = pd.concat(dfs, axis=1)\n    del dfs\n    import gc\n    gc.collect()\n    if verbose:\n        end_mem = df_out.memory_usage().sum() / 1024**2\n        num_reduction = str(100 * (start_mem - end_mem) / start_mem)\n        print(f'Mem. usage decreased to {str(end_mem)[:3]}Mb:  {num_reduction[:2]}% reduction')\n    return df_out\n\n\ndef reduce_mem_usage_inplace(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2\n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:47:24.604306Z","iopub.execute_input":"2022-07-09T11:47:24.60474Z","iopub.status.idle":"2022-07-09T11:47:24.620359Z","shell.execute_reply.started":"2022-07-09T11:47:24.604698Z","shell.execute_reply":"2022-07-09T11:47:24.618869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.データ確認","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:24:33.078555Z","iopub.execute_input":"2022-07-09T11:24:33.079016Z","iopub.status.idle":"2022-07-09T11:24:33.08424Z","shell.execute_reply.started":"2022-07-09T11:24:33.078961Z","shell.execute_reply":"2022-07-09T11:24:33.083102Z"}}},{"cell_type":"markdown","source":"## 1.1 特徴一覧\n日本語 <br>\nこのコンペティションの目的は、毎月の顧客プロファイルに基づいて、ある顧客が将来クレジットカードの残高を返さない確率を予測することです。対象の二項変数は、最新のクレジットカード明細書から18ヶ月間のパフォーマンスウィンドウを観察することによって計算され、もし顧客が最新の明細書の日付から120日以内に返済額を支払わない場合は、デフォルトイベントとみなされます。<br>\n\nデータセットには、各顧客の各明細書日付における集約されたプロファイル特徴が含まれています。特徴は匿名化、正規化されており、以下の一般的なカテゴリに分類されます。<br>\n\nD_* = 延滞変数<br>\nS_* = 支出変数<br>\nP_* = 支払い変数<br>\nB_* = 残高変数<br>\nR_* = リスク変数<br>\nであり、以下の特徴はカテゴリである。<br>\n[B_30', 'B_38', 'd_114', 'd_116', 'd_117', 'd_120', 'd_126', 'd_63', 'd_64', 'd_66', 'd_68' ]<br>\n\nあなたのタスクは、各顧客IDについて、将来の支払い不履行の確率を予測することです（ターゲット=1）。<br>\n\nこのデータセットでは、ネガティブ・クラスは5%でサブサンプルされているので、スコアリング・メトリックでは20倍の重み付けを受けることに注意してください。<br>\n\n\nEnglish<br>\nThe objective of this competition is to predict the probability that a customer does not pay back their credit card balance amount in the future based on their monthly customer profile. The target binary variable is calculated by observing 18 months performance window after the latest credit card statement, and if the customer does not pay due amount in 120 days after their latest statement date it is considered a default event.<br>\n\nThe dataset contains aggregated profile features for each customer at each statement date. Features are anonymized and normalized, and fall into the following general categories:<br>\n\nD_* = Delinquency variables<br>\nS_* = Spend variables<br>\nP_* = Payment variables<br>\nB_* = Balance variables<br>\nR_* = Risk variables<br>\nwith the following features being categorical:<br>\n\n['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']<br>\n\nYour task is to predict, for each customer_ID, the probability of a future payment default (target = 1).<br>\n\nNote that the negative class has been subsampled for this dataset at 5%, and thus receives a 20x weighting in the scoring metric.<br>","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"../input/amex-default-prediction/train_data.csv\", nrows=0)\nsample","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:47:24.622343Z","iopub.execute_input":"2022-07-09T11:47:24.623618Z","iopub.status.idle":"2022-07-09T11:47:24.71048Z","shell.execute_reply.started":"2022-07-09T11:47:24.623572Z","shell.execute_reply":"2022-07-09T11:47:24.709499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"../input/amex-default-prediction/train_data.csv\", nrows=5)\ncolumns = {\n    k: [c for c in sample.columns if \"{:}_\".format(k) in c] for k in [\"D\",\"S\", \"P\",\"B\",\"R\"]\n}\nprint(len(sample.columns))\nfor k,v in columns.items():\n    print(k, len(v))\nsample","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:47:31.622427Z","iopub.execute_input":"2022-07-09T11:47:31.6233Z","iopub.status.idle":"2022-07-09T11:47:31.677274Z","shell.execute_reply.started":"2022-07-09T11:47:31.623247Z","shell.execute_reply":"2022-07-09T11:47:31.675758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor k,v in columns.items():\n    train = pd.read_csv(\"../input/amex-default-prediction/train_data.csv\", usecols=[\"customer_ID\"]+v)\n    reduce_mem_usage(train)\n    train.to_feather(\"train_data_{:}.ftr\".format(k))\n# test = pd.read_csv(\"../input/amex-default-prediction/test_data.csv\")\n# labels = pd.read_csv(\"../input/amex-default-prediction/train_labels.csv\")\ntrain = pd.concat([pd.read_feather(\"train_data_{:}.ftr\".format(k)) for k in columns], axis=1)\ntrain = train[\"customer_ID\"].iloc[:,[0]].join(train.drop(\"customer_ID\", axis=1))\ntrain.to_feather(\"train_data.ftr\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:47:56.598586Z","iopub.execute_input":"2022-07-09T11:47:56.599061Z","iopub.status.idle":"2022-07-09T12:09:12.12102Z","shell.execute_reply.started":"2022-07-09T11:47:56.599021Z","shell.execute_reply":"2022-07-09T12:09:12.117743Z"},"trusted":true},"execution_count":null,"outputs":[]}]}