{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":31254,"databundleVersionId":3103714,"sourceType":"competition"},{"sourceId":87686672,"sourceType":"kernelVersion"},{"sourceId":87997705,"sourceType":"kernelVersion"}],"dockerImageVersionId":30162,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**I tried binary classification solution using LightGBM like CTR prediction**  \n**Please upvote if this notebook is useful!**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport gc\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nimport os\nimport joblib\nimport re\nimport tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2024-06-24T03:12:10.897521Z","iopub.execute_input":"2024-06-24T03:12:10.898238Z","iopub.status.idle":"2024-06-24T03:12:14.229377Z","shell.execute_reply.started":"2024-06-24T03:12:10.898138Z","shell.execute_reply":"2024-06-24T03:12:14.228715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Config:\n    transaction_path = \"../input/h-and-m-personalized-fashion-recommendations/transactions_train.csv\"\n    transaction_2020_path = \"../input/h-and-m-split-dataset-by-year/transactions_train_2020.csv\"\n    transaction_2019_path = \"../input/h-and-m-split-dataset-by-year/transactions_train_2019.csv\"\n    customer_path = \"../input/h-and-m-personalized-fashion-recommendations/customers.csv\"\n    article_path = \"../input/h-and-m-personalized-fashion-recommendations/articles.csv\"\n    image_feat_path = \"../input/h-and-m-swint-image-embedding/swin_tiny_patch4_window7_224_emb.csv.gz\"\n    sample_submission_path = \"../input/h-and-m-personalized-fashion-recommendations/sample_submission.csv\"\n\n    output_dir = \"../output/\"\n    #start_date = '2020-08-01'\n    start_date = '2020-09-15'\n\n    image_feat_dim = 768\n    text_feat_dim = 384\n    \n    #n_fold = 2\n    n_fold = 5\n    seed = 2022\n    lgbm = {\"n_estimators\" :50}\n\n    label = \"label\"\n\nos.makedirs(Config.output_dir, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:12:20.390452Z","iopub.execute_input":"2024-06-24T03:12:20.390759Z","iopub.status.idle":"2024-06-24T03:12:20.399018Z","shell.execute_reply.started":"2024-06-24T03:12:20.390706Z","shell.execute_reply":"2024-06-24T03:12:20.398078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/gemartin/load-data-reduce-memory-usage\ndef reduce_mem_usage(df):\n    \"\"\" iterate through all the columns of a dataframe and modify the data type\n        to reduce memory usage.        \n    \"\"\"\n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype\n        \n        if col_type != object:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)\n        else:\n            df[col] = df[col].astype('category')\n\n    end_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage after optimization is: {:.2f} MB'.format(end_mem))\n    print('Decreased by {:.1f}%'.format(100 * (start_mem - end_mem) / start_mem))\n    \n    return df\n\ndef import_data(file):\n    \"\"\"create a dataframe and optimize its memory usage\"\"\"\n    df = pd.read_csv(file, parse_dates=True, keep_date_col=True)\n    df = reduce_mem_usage(df)\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:12:44.979609Z","iopub.execute_input":"2024-06-24T03:12:44.979918Z","iopub.status.idle":"2024-06-24T03:12:44.997165Z","shell.execute_reply.started":"2024-06-24T03:12:44.979860Z","shell.execute_reply":"2024-06-24T03:12:44.996216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# preprae truth/false\n\nGiven transaction data, that are user-item pairs, are defined as positive data.  \nNegative data are created by shuffing user-item pairs.","metadata":{}},{"cell_type":"code","source":"df_trans = pd.read_csv(Config.transaction_2020_path)\ndf_trans = df_trans[df_trans[\"t_dat\"] >= Config.start_date].reset_index(drop=True)\ndf_trans = reduce_mem_usage(df_trans)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:10.974532Z","iopub.execute_input":"2024-06-24T03:14:10.975139Z","iopub.status.idle":"2024-06-24T03:14:25.641394Z","shell.execute_reply.started":"2024-06-24T03:14:10.975104Z","shell.execute_reply":"2024-06-24T03:14:25.640537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_truth = df_trans[[\"customer_id\", \"article_id\"]]\ndf_truth.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:25.643021Z","iopub.execute_input":"2024-06-24T03:14:25.643309Z","iopub.status.idle":"2024-06-24T03:14:25.657460Z","shell.execute_reply.started":"2024-06-24T03:14:25.643271Z","shell.execute_reply":"2024-06-24T03:14:25.656669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_trans","metadata":{"execution":{"iopub.status.busy":"2024-06-23T07:29:14.593356Z","iopub.execute_input":"2024-06-23T07:29:14.593640Z","iopub.status.idle":"2024-06-23T07:29:14.597612Z","shell.execute_reply.started":"2024-06-23T07:29:14.593606Z","shell.execute_reply":"2024-06-23T07:29:14.596748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_false = df_truth.copy()\ndf_false.loc[:, \"article_id\"] = df_false[\"article_id\"].sample(frac=1).tolist()\ndf_false.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:35.498934Z","iopub.execute_input":"2024-06-24T03:14:35.499580Z","iopub.status.idle":"2024-06-24T03:14:35.663535Z","shell.execute_reply.started":"2024-06-24T03:14:35.499540Z","shell.execute_reply":"2024-06-24T03:14:35.662838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_truth.loc[:, Config.label] = 1\ndf_false.loc[:, Config.label] = 0","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:43.380529Z","iopub.execute_input":"2024-06-24T03:14:43.381106Z","iopub.status.idle":"2024-06-24T03:14:43.387425Z","shell.execute_reply.started":"2024-06-24T03:14:43.381069Z","shell.execute_reply":"2024-06-24T03:14:43.386685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_truth = pd.concat([df_truth, df_false])\ndf_truth.shape, ","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:46.938090Z","iopub.execute_input":"2024-06-24T03:14:46.938666Z","iopub.status.idle":"2024-06-24T03:14:46.989605Z","shell.execute_reply.started":"2024-06-24T03:14:46.938627Z","shell.execute_reply":"2024-06-24T03:14:46.988917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_truth[df_truth[\"label\"] ==1].shape,  df_truth[df_truth[\"label\"] ==0].shape, ","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:53.606515Z","iopub.execute_input":"2024-06-24T03:14:53.607157Z","iopub.status.idle":"2024-06-24T03:14:53.633352Z","shell.execute_reply.started":"2024-06-24T03:14:53.607122Z","shell.execute_reply":"2024-06-24T03:14:53.632636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_false","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:55.186639Z","iopub.execute_input":"2024-06-24T03:14:55.187398Z","iopub.status.idle":"2024-06-24T03:14:55.190707Z","shell.execute_reply.started":"2024-06-24T03:14:55.187359Z","shell.execute_reply":"2024-06-24T03:14:55.189994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing\n## prepare article feat","metadata":{}},{"cell_type":"code","source":"df_article = import_data(Config.article_path)\n#df_image = import_data(Config.image_feat_path)\n#df_text = import_data(Config.text_feat_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:14:58.016648Z","iopub.execute_input":"2024-06-24T03:14:58.017496Z","iopub.status.idle":"2024-06-24T03:14:59.378057Z","shell.execute_reply.started":"2024-06-24T03:14:58.017449Z","shell.execute_reply":"2024-06-24T03:14:59.377234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_table_feat(df):\n        \n    article_id_cols = [\"product_code\", \"product_type_no\", \"graphical_appearance_no\", \"colour_group_code\",\n              \"perceived_colour_value_id\", \"perceived_colour_master_id\", \"department_no\", \"index_group_no\", \n               \"section_no\", \"garment_group_no\"]\n    \n    article_dummy_cols = [\"product_type_name\", \"product_group_name\", \"graphical_appearance_name\", \"colour_group_name\",\n                         \"perceived_colour_value_name\", \"perceived_colour_master_name\", \n                         #\"department_name\", \n                         \"index_name\", \"index_group_name\", \"section_name\", \"garment_group_name\"]\n    \n    article_drop_cols = [\"index_code\", \"prod_name\", \"detail_desc\", \"department_name\"]\n    \n    df = df.drop(article_drop_cols, axis=1)\n    df = pd.get_dummies(df, columns=article_dummy_cols)\n    return df\n    \ndef create_article_feat(df_article, \n                        #df_image\n                        ):\n\n    # rename image \n    #rename_dic = {f\"{i}\": f\"image_col_{i}\" for i in range(Config.image_feat_dim)}\n    #df_image = df_image.rename(columns=rename_dic)\n\n    df_article_feat = get_table_feat(df_article)\n    #df_article_feat = df_article_feat.merge(df_image, on=\"article_id\", how=\"left\")\n\n    return df_article_feat","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:15:01.831227Z","iopub.execute_input":"2024-06-24T03:15:01.831520Z","iopub.status.idle":"2024-06-24T03:15:01.838858Z","shell.execute_reply.started":"2024-06-24T03:15:01.831486Z","shell.execute_reply":"2024-06-24T03:15:01.838100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#df_article_feat = create_article_feat(df_article, df_image, df_text)\ndf_article_feat = create_article_feat(df_article)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:15:05.185372Z","iopub.execute_input":"2024-06-24T03:15:05.186135Z","iopub.status.idle":"2024-06-24T03:15:05.415849Z","shell.execute_reply.started":"2024-06-24T03:15:05.186093Z","shell.execute_reply":"2024-06-24T03:15:05.415028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_article_feat.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:15:06.263613Z","iopub.execute_input":"2024-06-24T03:15:06.264001Z","iopub.status.idle":"2024-06-24T03:15:06.283093Z","shell.execute_reply.started":"2024-06-24T03:15:06.263957Z","shell.execute_reply":"2024-06-24T03:15:06.282411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del df_article, df_image, df_text\ndel df_article","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:15:08.884381Z","iopub.execute_input":"2024-06-24T03:15:08.884683Z","iopub.status.idle":"2024-06-24T03:15:08.893587Z","shell.execute_reply.started":"2024-06-24T03:15:08.884649Z","shell.execute_reply":"2024-06-24T03:15:08.892628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_article_feat = reduce_mem_usage(df_article_feat)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:15:09.210327Z","iopub.execute_input":"2024-06-24T03:15:09.210915Z","iopub.status.idle":"2024-06-24T03:15:09.871681Z","shell.execute_reply.started":"2024-06-24T03:15:09.210865Z","shell.execute_reply":"2024-06-24T03:15:09.870923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## prepare customer feat","metadata":{}},{"cell_type":"code","source":"df_customer = pd.read_csv(Config.customer_path)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:17:27.497171Z","iopub.execute_input":"2024-06-24T03:17:27.497490Z","iopub.status.idle":"2024-06-24T03:17:32.660041Z","shell.execute_reply.started":"2024-06-24T03:17:27.497456Z","shell.execute_reply":"2024-06-24T03:17:32.659333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_customer_feat(df):\n    \n    customer_drop_cols = [\"postal_code\"]\n    customer_dummy_cols = [\"club_member_status\", \"fashion_news_frequency\"]\n    \n    \n    df = df.drop(customer_drop_cols, axis=1)\n    df.loc[:, \"FN\"] = df[\"FN\"].fillna(0)\n    df.loc[:, \"Active\"] = df[\"Active\"].fillna(0)\n    df.loc[:, \"club_member_status\"] = df[\"club_member_status\"].fillna(\"NONE\")\n    df.loc[:, \"fashion_news_frequency\"] = df[\"fashion_news_frequency\"].fillna(\"NONE\")\n    df.loc[:, \"age\"] = df[\"age\"].fillna(0)\n    df.loc[:, \"age\"] = np.log1p(df[\"age\"])\n\n    df = pd.get_dummies(df, columns=customer_dummy_cols)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:17:32.661382Z","iopub.execute_input":"2024-06-24T03:17:32.661608Z","iopub.status.idle":"2024-06-24T03:17:32.669484Z","shell.execute_reply.started":"2024-06-24T03:17:32.661579Z","shell.execute_reply":"2024-06-24T03:17:32.668561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_customer_feat = create_customer_feat(df_customer)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:17:32.670665Z","iopub.execute_input":"2024-06-24T03:17:32.671008Z","iopub.status.idle":"2024-06-24T03:17:33.649116Z","shell.execute_reply.started":"2024-06-24T03:17:32.670973Z","shell.execute_reply":"2024-06-24T03:17:33.648296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_customer\ndf_customer_feat = reduce_mem_usage(df_customer_feat)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:17:33.650678Z","iopub.execute_input":"2024-06-24T03:17:33.650894Z","iopub.status.idle":"2024-06-24T03:17:36.241574Z","shell.execute_reply.started":"2024-06-24T03:17:33.650852Z","shell.execute_reply":"2024-06-24T03:17:36.240742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge all feats and construct dataset","metadata":{}},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:17:38.725158Z","iopub.execute_input":"2024-06-24T03:17:38.725422Z","iopub.status.idle":"2024-06-24T03:17:38.872831Z","shell.execute_reply.started":"2024-06-24T03:17:38.725393Z","shell.execute_reply":"2024-06-24T03:17:38.872066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" # https://github.com/awslabs/autogluon/issues/399\ndf_article_feat = df_article_feat.rename(columns = lambda x:re.sub('[^A-Za-z0-9_]+', '', x))\ndf_customer_feat = df_customer_feat.rename(columns = lambda x:re.sub('[^A-Za-z0-9_]+', '', x))","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:25:46.114518Z","iopub.execute_input":"2024-06-24T03:25:46.115143Z","iopub.status.idle":"2024-06-24T03:25:46.159970Z","shell.execute_reply.started":"2024-06-24T03:25:46.115103Z","shell.execute_reply":"2024-06-24T03:25:46.159291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_dataset(df_truth, df_article_feat, df_customer_feat):\n\n#     df_data = df_truth.merge(df_article_feat, on=\"article_id\", how='left')\n#     df_data = df_data.merge(df_customer_feat, on = \"customer_id\", how='left')    \n#     df_data = df_data.drop([\"customer_id\", \"article_id\"], axis=1)    \n#     df_data = df_data.fillna(0)\n\n#     # https://github.com/awslabs/autogluon/issues/399\n#     df_data = df_data.rename(columns = lambda x:re.sub('[^A-Za-z0-9_]+', '', x))\n\n#     return df_data","metadata":{"execution":{"iopub.status.busy":"2022-02-26T05:46:51.581118Z","iopub.execute_input":"2022-02-26T05:46:51.581423Z","iopub.status.idle":"2022-02-26T05:46:51.586482Z","shell.execute_reply.started":"2022-02-26T05:46:51.581392Z","shell.execute_reply":"2022-02-26T05:46:51.585254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_article_feat","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:25:53.085029Z","iopub.execute_input":"2024-06-24T03:25:53.085652Z","iopub.status.idle":"2024-06-24T03:25:53.128202Z","shell.execute_reply.started":"2024-06-24T03:25:53.085615Z","shell.execute_reply":"2024-06-24T03:25:53.127532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/tkm2261/fast-pandas-left-join-357x-faster-than-pd-merge\n\ndf_article_feat = df_article_feat.set_index(\"article_id\")\ndf_customer_feat = df_customer_feat.set_index(\"customer_id\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:06.667925Z","iopub.execute_input":"2024-06-24T03:26:06.668641Z","iopub.status.idle":"2024-06-24T03:26:06.693819Z","shell.execute_reply.started":"2024-06-24T03:26:06.668603Z","shell.execute_reply":"2024-06-24T03:26:06.692683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset_faster(df_truth, df_article_feat, df_customer_feat): \n\n    df_data = pd.concat([\n        df_truth.reset_index(drop=True), \n        df_article_feat.reindex(df_truth['article_id'].values).reset_index(drop=True)\n    ], axis=1)\n    df_data = pd.concat([\n        df_data.reset_index(drop=True), \n        df_customer_feat.reindex(df_data['customer_id'].values).reset_index(drop=True)\n    ], axis=1)  \n    \n    df_data = df_data.drop([\"customer_id\", \"article_id\"], axis=1)    \n    df_data = df_data.fillna(0)\n\n\n    return df_data","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:06.879749Z","iopub.execute_input":"2024-06-24T03:26:06.880528Z","iopub.status.idle":"2024-06-24T03:26:06.886896Z","shell.execute_reply.started":"2024-06-24T03:26:06.880487Z","shell.execute_reply":"2024-06-24T03:26:06.886058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data = create_dataset_faster(df_truth, df_article_feat, df_customer_feat)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:07.238401Z","iopub.execute_input":"2024-06-24T03:26:07.238671Z","iopub.status.idle":"2024-06-24T03:26:14.705408Z","shell.execute_reply.started":"2024-06-24T03:26:07.238642Z","shell.execute_reply":"2024-06-24T03:26:14.704723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.to_pickle(f\"{Config.output_dir}/feature.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:27.094079Z","iopub.execute_input":"2024-06-24T03:26:27.094367Z","iopub.status.idle":"2024-06-24T03:26:28.344519Z","shell.execute_reply.started":"2024-06-24T03:26:27.094335Z","shell.execute_reply":"2024-06-24T03:26:28.343711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.columns.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:28.345997Z","iopub.execute_input":"2024-06-24T03:26:28.346602Z","iopub.status.idle":"2024-06-24T03:26:28.352021Z","shell.execute_reply.started":"2024-06-24T03:26:28.346559Z","shell.execute_reply":"2024-06-24T03:26:28.351345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data[df_data[\"label\"] ==1].shape,  df_data[df_data[\"label\"] ==0].shape, ","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:30.738637Z","iopub.execute_input":"2024-06-24T03:26:30.739298Z","iopub.status.idle":"2024-06-24T03:26:32.277138Z","shell.execute_reply.started":"2024-06-24T03:26:30.739262Z","shell.execute_reply":"2024-06-24T03:26:32.276180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:32.278643Z","iopub.execute_input":"2024-06-24T03:26:32.278869Z","iopub.status.idle":"2024-06-24T03:26:32.348668Z","shell.execute_reply.started":"2024-06-24T03:26:32.278841Z","shell.execute_reply":"2024-06-24T03:26:32.347853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"def train(df_data):    \n    cols = [col for col in df_data.columns if Config.label != col]\n\n    folds = StratifiedKFold(n_splits=Config.n_fold, random_state=Config.seed, shuffle=True)\n    es = early_stopping(1000)\n    le = log_evaluation(period=100)\n    scores = []    \n        \n    for fold, (train_idx, val_idx) in enumerate(folds.split(df_data, df_data[Config.label])):\n        print(f\"=====fold {fold}=======\")\n\n        df_train = df_data.loc[train_idx].reset_index(drop=True)\n        df_val = df_data.loc[val_idx].reset_index(drop=True)\n        \n        print(\"train shape\", df_train.shape, \"test shape\", df_val.shape)\n        \n        model = LGBMClassifier(random_state=Config.seed, **Config.lgbm)\n        \n        model.fit(df_train[cols], df_train[Config.label],\n                eval_set=(df_val[cols], df_val[Config.label]),\n                callbacks=[es, le],\n                eval_metric=\"auc\"              \n                )\n        \n        # validation\n        val_pred = model.predict(df_val[cols])\n        val_score = roc_auc_score(df_val[Config.label], val_pred)\n        scores.append(val_score)\n        \n        # save_model\n        joblib.dump(model,f\"lgbm_fold_{fold}.joblib\")\n\n    return scores","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:35.612794Z","iopub.execute_input":"2024-06-24T03:26:35.613518Z","iopub.status.idle":"2024-06-24T03:26:35.623734Z","shell.execute_reply.started":"2024-06-24T03:26:35.613482Z","shell.execute_reply":"2024-06-24T03:26:35.622908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = train(df_data)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:26:36.695746Z","iopub.execute_input":"2024-06-24T03:26:36.696512Z","iopub.status.idle":"2024-06-24T03:27:16.958240Z","shell.execute_reply.started":"2024-06-24T03:26:36.696471Z","shell.execute_reply":"2024-06-24T03:27:16.957542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(scores)\nprint(np.mean(scores))","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:27:27.185306Z","iopub.execute_input":"2024-06-24T03:27:27.185902Z","iopub.status.idle":"2024-06-24T03:27:27.190655Z","shell.execute_reply.started":"2024-06-24T03:27:27.185852Z","shell.execute_reply":"2024-06-24T03:27:27.189825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-28T04:52:56.769231Z","iopub.execute_input":"2023-11-28T04:52:56.769561Z","iopub.status.idle":"2023-11-28T04:52:56.777132Z","shell.execute_reply.started":"2023-11-28T04:52:56.769526Z","shell.execute_reply":"2023-11-28T04:52:56.776076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature importance","metadata":{}},{"cell_type":"code","source":"def get_feat_imp(df_data):\n    imps_list = []    \n    cols = [col for col in df_data.columns if Config.label != col]\n    for _fold in range(Config.n_fold):\n        with open(f\"lgbm_fold_{_fold}.joblib\", \"rb\") as f:\n            model = joblib.load(f)\n        imps= model.feature_importances_\n        imps_list.append(imps)\n\n    imps = np.mean(imps_list, axis=0)\n    df_imps = pd.DataFrame({\"columns\": df_data[cols].columns.tolist(), \"feat_imp\": imps})\n    df_imps = df_imps.sort_values(\"feat_imp\", ascending=False).reset_index(drop=True)\n\n    return df_imps\n ","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:28:29.111854Z","iopub.execute_input":"2024-06-24T03:28:29.112662Z","iopub.status.idle":"2024-06-24T03:28:29.119649Z","shell.execute_reply.started":"2024-06-24T03:28:29.112619Z","shell.execute_reply":"2024-06-24T03:28:29.118897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_fea_imp = get_feat_imp(df_data)\ndf_fea_imp.head(30)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:28:37.538446Z","iopub.execute_input":"2024-06-24T03:28:37.538962Z","iopub.status.idle":"2024-06-24T03:28:38.225234Z","shell.execute_reply.started":"2024-06-24T03:28:37.538927Z","shell.execute_reply":"2024-06-24T03:28:38.224498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(10, 8))\nsns.barplot(data=df_fea_imp.head(30), x=\"feat_imp\", y=\"columns\")","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:28:52.660921Z","iopub.execute_input":"2024-06-24T03:28:52.661620Z","iopub.status.idle":"2024-06-24T03:28:53.176950Z","shell.execute_reply.started":"2024-06-24T03:28:52.661571Z","shell.execute_reply":"2024-06-24T03:28:53.176206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference (only 10 samples)","metadata":{}},{"cell_type":"code","source":"df_submission = import_data(Config.sample_submission_path)\ndf_submission.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:30:07.872294Z","iopub.execute_input":"2024-06-24T03:30:07.873059Z","iopub.status.idle":"2024-06-24T03:30:13.060099Z","shell.execute_reply.started":"2024-06-24T03:30:07.873019Z","shell.execute_reply":"2024-06-24T03:30:13.059384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_submission.iloc[0, 1].split(\" \"))","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:30:35.600254Z","iopub.execute_input":"2024-06-24T03:30:35.600545Z","iopub.status.idle":"2024-06-24T03:30:35.606574Z","shell.execute_reply.started":"2024-06-24T03:30:35.600510Z","shell.execute_reply":"2024-06-24T03:30:35.605773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:30:39.224536Z","iopub.execute_input":"2024-06-24T03:30:39.224820Z","iopub.status.idle":"2024-06-24T03:30:39.231788Z","shell.execute_reply.started":"2024-06-24T03:30:39.224788Z","shell.execute_reply":"2024-06-24T03:30:39.231009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_article = import_data(Config.article_path)\ndf_article = df_article[[\"article_id\"]]","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:29:07.984945Z","iopub.execute_input":"2024-06-24T03:29:07.985469Z","iopub.status.idle":"2024-06-24T03:29:08.922664Z","shell.execute_reply.started":"2024-06-24T03:29:07.985429Z","shell.execute_reply":"2024-06-24T03:29:08.921908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_article.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:29:11.671148Z","iopub.execute_input":"2024-06-24T03:29:11.671439Z","iopub.status.idle":"2024-06-24T03:29:11.679313Z","shell.execute_reply.started":"2024-06-24T03:29:11.671407Z","shell.execute_reply":"2024-06-24T03:29:11.678531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef inference(df_submission, df_article, df_article_feat, df_customer_feat, models, cols):\n\n    article_candidates = []\n\n    for customer in tqdm.tqdm(df_submission[\"customer_id\"]):\n        _df = df_article.copy()\n        _df.loc[:, \"customer_id\"] = customer\n        _df = create_dataset_faster(_df, df_article_feat, df_customer_feat)         \n        _df = _df[cols]     \n\n        preds = []\n        for _fold in range(Config.n_fold):\n            pred = models[_fold].predict_proba(_df, num_iteration=models[_fold]._best_iteration)[:, 1]\n            preds.append(pred)\n        \n        pred = np.mean(preds, axis=0)        \n        df_pred = pd.DataFrame({\"article_id\": df_article[\"article_id\"].tolist() , \"score\": pred})\n                \n        df_pred = df_pred.sort_values(\"score\", ascending=False).reset_index(drop=True)\n        df_pred = df_pred.head(12)\n        pred_str = [str(pred) for pred in df_pred[\"article_id\"].tolist()]\n        article_candidates.append(\" \".join(pred_str))\n\n    df_submission.loc[:, \"prediction\"] = article_candidates\n\n    return df_submission","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:31:03.626687Z","iopub.execute_input":"2024-06-24T03:31:03.627326Z","iopub.status.idle":"2024-06-24T03:31:03.637198Z","shell.execute_reply.started":"2024-06-24T03:31:03.627285Z","shell.execute_reply":"2024-06-24T03:31:03.636381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predict only 10 samples.  \nIt requires a lot of time to predict all data, 😥**","metadata":{}},{"cell_type":"code","source":"df_article_feat","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:32:54.995096Z","iopub.execute_input":"2024-06-24T03:32:54.995838Z","iopub.status.idle":"2024-06-24T03:32:55.037411Z","shell.execute_reply.started":"2024-06-24T03:32:54.995797Z","shell.execute_reply":"2024-06-24T03:32:55.036663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models = []\nfor _fold in range(Config.n_fold):\n    with open(f\"lgbm_fold_{_fold}.joblib\", \"rb\") as f:\n        model = joblib.load(f)\n        models.append(model)\n    \ncols = [col for col in df_data.columns if Config.label != col]\ndf_sub = inference(df_submission.iloc[:10], df_article.iloc[:10], df_article_feat.iloc[:10], df_customer_feat.iloc[:10], models, cols)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:33:13.009749Z","iopub.execute_input":"2024-06-24T03:33:13.010042Z","iopub.status.idle":"2024-06-24T03:33:20.441467Z","shell.execute_reply.started":"2024-06-24T03:33:13.010011Z","shell.execute_reply":"2024-06-24T03:33:20.440792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:33:21.327868Z","iopub.execute_input":"2024-06-24T03:33:21.328616Z","iopub.status.idle":"2024-06-24T03:33:21.338160Z","shell.execute_reply.started":"2024-06-24T03:33:21.328575Z","shell.execute_reply":"2024-06-24T03:33:21.337325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\", index=None)","metadata":{"execution":{"iopub.status.busy":"2024-06-24T03:33:26.345305Z","iopub.execute_input":"2024-06-24T03:33:26.346102Z","iopub.status.idle":"2024-06-24T03:33:26.386621Z","shell.execute_reply.started":"2024-06-24T03:33:26.346042Z","shell.execute_reply":"2024-06-24T03:33:26.386047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}