{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ninput_dir = '../input/amex-default-prediction'\nparquet_dir = '../input/amex-data-integer-dtypes-parquet-format'\noutput_dir = '../output/2'","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:02:36.127282Z","iopub.status.busy":"2022-08-02T09:02:36.126394Z","iopub.status.idle":"2022-08-02T09:02:36.155390Z","shell.execute_reply":"2022-08-02T09:02:36.154273Z","shell.execute_reply.started":"2022-08-02T09:02:36.127160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:56:32.779711Z","iopub.status.busy":"2022-08-02T09:56:32.779309Z","iopub.status.idle":"2022-08-02T09:56:32.818059Z","shell.execute_reply":"2022-08-02T09:56:32.817112Z","shell.execute_reply.started":"2022-08-02T09:56:32.779679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(f'{parquet_dir}/train.parquet')\ntrain_labels = pd.read_csv(f'{input_dir}/train_labels.csv')\nsubmission = pd.read_csv(f'{input_dir}/sample_submission.csv')","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:02:36.158551Z","iopub.status.busy":"2022-08-02T09:02:36.158192Z","iopub.status.idle":"2022-08-02T09:03:05.603936Z","shell.execute_reply":"2022-08-02T09:03:05.601509Z","shell.execute_reply.started":"2022-08-02T09:02:36.158520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:03:05.606970Z","iopub.status.busy":"2022-08-02T09:03:05.606572Z","iopub.status.idle":"2022-08-02T09:03:08.078877Z","shell.execute_reply":"2022-08-02T09:03:08.077682Z","shell.execute_reply.started":"2022-08-02T09:03:05.606934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_timestamp = train[['customer_ID', 'S_2']].groupby('customer_ID').agg(['count', 'max']).reset_index()\ntrain_timestamp_columns = [i[0]+i[1] for i in train_timestamp.columns]\ntrain_timestamp.columns = train_timestamp_columns\ntrain_timestamp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータと合わせるために、①countの正規化、②maxを「最新データからの乖離」の指標に変換\n# countの正規化\nfrom sklearn.preprocessing import StandardScaler\nss = StandardScaler()\ntrain_timestamp['S_2count_standard'] = ss.fit_transform(pd.DataFrame(train_timestamp['S_2count']))\n# 2.追加 最大数レコードが存在するかそうでないかのフラグを付与\ntrain_timestamp['record_max_exist'] = np.where(train_timestamp['S_2count']==train_timestamp['S_2count'].max(), 1, 0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_timestamp['S_2count'].unique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 最新データからの乖離を作ろうとしたが、みんな変わらなかった\ntrain_timestamp['max_nengetsu'] = train_timestamp['S_2max'].str[0:4] + train_timestamp['S_2max'].str[5:7]\ntrain_timestamp['max_nengetsu'].unique()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_timestamp","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# groupbyで平均、合計、最初の数値、最後の数値を算出\ntrain_sum = train.groupby('customer_ID').agg(['mean', 'sum', 'first', 'last', 'min', 'max', 'std']).reset_index()\ntrain_sum_columns = [i[0]+i[1] for i in train_sum.columns]\ntrain_sum.columns = train_sum_columns\ntrain_sum","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:03:10.935619Z","iopub.status.busy":"2022-08-02T09:03:10.935274Z","iopub.status.idle":"2022-08-02T09:03:46.927532Z","shell.execute_reply":"2022-08-02T09:03:46.926359Z","shell.execute_reply.started":"2022-08-02T09:03:10.935588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2追加. lastとfirstの差/比率、lastと平均の差/比率を特徴量として付与\nfeature_cols = [col for col in train.columns if col not in ('customer_ID', 'S_2')]\nfor i in feature_cols:\n    train_sum[f\"{i}_last-first\"] = train_sum[f\"{i}last\"] - train_sum[f\"{i}first\"]\n    train_sum[f\"{i}_last/first\"] = train_sum[f\"{i}last\"] / train_sum[f\"{i}first\"]\n    train_sum[f\"{i}_last-mean\"] = train_sum[f\"{i}last\"] - train_sum[f\"{i}mean\"]\n    train_sum[f\"{i}_last/mean\"] = train_sum[f\"{i}last\"] / train_sum[f\"{i}mean\"]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_labels.merge(train_sum, how='left', on='customer_ID')\ntrain = train.merge(train_timestamp[['customer_ID', 'S_2count_standard', 'record_max_exist']], how='left', on='customer_ID')","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:09:59.113432Z","iopub.status.busy":"2022-08-02T09:09:59.112730Z","iopub.status.idle":"2022-08-02T09:10:16.121638Z","shell.execute_reply":"2022-08-02T09:10:16.120324Z","shell.execute_reply.started":"2022-08-02T09:09:59.113390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:11:32.114710Z","iopub.status.busy":"2022-08-02T09:11:32.113680Z","iopub.status.idle":"2022-08-02T09:11:32.384576Z","shell.execute_reply":"2022-08-02T09:11:32.383392Z","shell.execute_reply.started":"2022-08-02T09:11:32.114664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = train.drop(columns=['target', 'customer_ID'])\ny = train['target']\n\nX_train, X_valid, y_train, y_valid = train_test_split(\n    X, y, test_size=0.2, random_state=0\n)","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:48:14.631418Z","iopub.status.busy":"2022-08-02T09:48:14.630961Z","iopub.status.idle":"2022-08-02T09:48:16.206113Z","shell.execute_reply":"2022-08-02T09:48:16.204733Z","shell.execute_reply.started":"2022-08-02T09:48:14.631383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, X, y","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:48:27.092103Z","iopub.status.busy":"2022-08-02T09:48:27.091604Z","iopub.status.idle":"2022-08-02T09:48:27.107333Z","shell.execute_reply":"2022-08-02T09:48:27.105663Z","shell.execute_reply.started":"2022-08-02T09:48:27.092063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 参考：https://blog.amedama.jp/entry/2018/05/01/081842\n# lightgbmをインポートしましょう\nimport lightgbm as lgb\n\n# データセットを生成しましょう\nlgb_train = lgb.Dataset(X_train, y_train)\nlgb_valid = lgb.Dataset(X_valid, y_valid, reference=lgb_train)\n\n# lightGBMのパラメータを辞書型で定義しましょう（参考：https://qiita.com/nabenabe0928/items/6b9772131ba89da00354）\n# lightGBMはパラメータ調整しなくても割といいスコアが出るので、「こんな書き方をするんだ」くらいに知っておけばOKです\nlgb_params = {\n    'objective': 'binary', \n    'metric': 'binary_logloss', \n    'boosting': 'dart', \n    'verbosity': -1, \n    'seed': 0\n}\n\n# lightGBMを学習しましょう(early_stoppingをかけるようにコードを書いてみてください)\nmodel = lgb.train(lgb_params, lgb_train, valid_sets=lgb_valid, num_boost_round=1000, verbose_eval=100)","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:51:44.699475Z","iopub.status.busy":"2022-08-02T09:51:44.698625Z","iopub.status.idle":"2022-08-02T09:52:25.363993Z","shell.execute_reply":"2022-08-02T09:52:25.362794Z","shell.execute_reply.started":"2022-08-02T09:51:44.699419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_true = pd.DataFrame(y_valid)\ny_pred = y_true.rename(columns={'target': 'prediction'})\ny_pred['prediction'] = model.predict(X_valid)","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:56:13.955286Z","iopub.status.busy":"2022-08-02T09:56:13.954120Z","iopub.status.idle":"2022-08-02T09:56:14.622197Z","shell.execute_reply":"2022-08-02T09:56:14.621194Z","shell.execute_reply.started":"2022-08-02T09:56:13.955241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# バリデーションスコアの算出\namex_metric(y_true, y_pred)","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:56:39.577808Z","iopub.status.busy":"2022-08-02T09:56:39.577362Z","iopub.status.idle":"2022-08-02T09:56:39.746316Z","shell.execute_reply":"2022-08-02T09:56:39.745163Z","shell.execute_reply.started":"2022-08-02T09:56:39.577772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y_true, y_pred, X_train, y_train, X_valid, y_valid","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:57:37.791358Z","iopub.status.busy":"2022-08-02T09:57:37.790867Z","iopub.status.idle":"2022-08-02T09:57:37.797292Z","shell.execute_reply":"2022-08-02T09:57:37.796131Z","shell.execute_reply.started":"2022-08-02T09:57:37.791313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')","metadata":{"execution":{"iopub.execute_input":"2022-08-02T09:58:22.283990Z","iopub.status.busy":"2022-08-02T09:58:22.282706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_timestamp = test[['customer_ID', 'S_2']].groupby('customer_ID').count().reset_index()\ntest_timestamp = test_timestamp.rename(columns={'S_2': 'S_2count_standard'})\ntest_timestamp['record_max_exist'] = np.where(test_timestamp['S_2count_standard']==test_timestamp['S_2count_standard'].max(), 1, 0)\nss = StandardScaler()\ntest_timestamp['S_2count_standard'] = ss.fit_transform(pd.DataFrame(test_timestamp['S_2count_standard']))\n\ntest_sum = test.groupby('customer_ID').agg(['mean', 'sum', 'first', 'last', 'min', 'max', 'std']).reset_index()\ntest_sum_columns = [i[0]+i[1] for i in test_sum.columns]\ntest_sum.columns = test_sum_columns\n\n# 2追加. lastとfirstの差/比率、lastと平均の差/比率を特徴量として付与\nfeature_cols = [col for col in test.columns if col not in ('customer_ID', 'S_2')]\nfor i in feature_cols:\n    test_sum[f\"{i}_last-first\"] = test_sum[f\"{i}last\"] - test_sum[f\"{i}first\"]\n    test_sum[f\"{i}_last/first\"] = test_sum[f\"{i}last\"] / test_sum[f\"{i}first\"]\n    test_sum[f\"{i}_last-mean\"] = test_sum[f\"{i}last\"] - test_sum[f\"{i}mean\"]\n    test_sum[f\"{i}_last/mean\"] = test_sum[f\"{i}last\"] / test_sum[f\"{i}mean\"]\n\ntest = test_sum.merge(test_timestamp, how='left', on='customer_ID')\n\ndel test_timestamp, test_sum\ntest","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_array = np.array([])\nprediction_array = np.concatenate([prediction_array, model.predict(test.iloc[:300000].drop(columns=['customer_ID']))])\nprint('step1 finished')\nprediction_array = np.concatenate([prediction_array, model.predict(test.iloc[300000:600000].drop(columns=['customer_ID']))])\nprint('step2 finished')\nprediction_array = np.concatenate([prediction_array, model.predict(test.iloc[600000:800000].drop(columns=['customer_ID']))])\nprint('step3 finished')\nprediction_array = np.concatenate([prediction_array, model.predict(test.iloc[800000:].drop(columns=['customer_ID']))])\nprint('step4 finished')\nsubmission['prediction'] = prediction_array\nsubmission","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pickle\nos.makedirs(output_dir, exist_ok=True)\nsubmission.to_csv(f'{output_dir}/submission.csv', index=False)\n\nwith open(f'{output_dir}/model.pickle', 'wb') as f:\n    pickle.dump(model, f)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}