{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## First of all I loaded all the data into Snowflake database for easier manipulation\nI made a simple datamart with snap dates '2020-09-01', '2020-09-08' for training\nand '2020-09-15' for testing, datamart was exported in one csv file named\nmodel_build_base.csv\n\n#### The train set contains following data:\n* 495,774 rows of SOLD=1 -> All transactions that were sold in week '2020-09-02' to '2020-09-08' and week '2020-09-09' to '2020-09-15'\n* 4,000,000 randomly chosen combinations of customer_id and article_id that were not sold with SOLD=0\n* All article information, customer information and customer transaction history info is also joined\n\n#### The test set contains following data:\n* Similar to train set just shifted forward for week between '2020-09-15' and '2020-09-22'\n\n#### Predictors\nPredictors were basic aggregations like number bought last month or days since \nlast purchase on three different levels:\n* CUST - on customer level, here I included also number of purchases across different product groups\n* ART - on article (item) level\n* CUSTART - combination of customer and article, last purchase, number purchased last month etc.\n\n\n#### Fitting\nA simple XGBoost model was fitted\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-06T13:01:16.795595Z","iopub.execute_input":"2022-11-06T13:01:16.797345Z","iopub.status.idle":"2022-11-06T13:01:16.804791Z","shell.execute_reply.started":"2022-11-06T13:01:16.797256Z","shell.execute_reply":"2022-11-06T13:01:16.803171Z"}}},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:36:08.237992Z","iopub.execute_input":"2022-11-13T15:36:08.239012Z","iopub.status.idle":"2022-11-13T15:36:08.270733Z","shell.execute_reply.started":"2022-11-13T15:36:08.238892Z","shell.execute_reply":"2022-11-13T15:36:08.269571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing custom library\n!pip install git+https://github.com/Vrboska/mofr@master","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-13T15:36:08.272589Z","iopub.execute_input":"2022-11-13T15:36:08.272942Z","iopub.status.idle":"2022-11-13T15:36:25.557060Z","shell.execute_reply.started":"2022-11-13T15:36:08.272911Z","shell.execute_reply":"2022-11-13T15:36:25.555606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport random\nimport mofr\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom xgboost import XGBClassifier, plot_tree\nimport math\n\nimport xgboost as xgb","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:36:25.558554Z","iopub.execute_input":"2022-11-13T15:36:25.558902Z","iopub.status.idle":"2022-11-13T15:36:26.506218Z","shell.execute_reply.started":"2022-11-13T15:36:25.558867Z","shell.execute_reply":"2022-11-13T15:36:26.504693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed=1234","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:36:26.508808Z","iopub.execute_input":"2022-11-13T15:36:26.509259Z","iopub.status.idle":"2022-11-13T15:36:26.514430Z","shell.execute_reply.started":"2022-11-13T15:36:26.509219Z","shell.execute_reply":"2022-11-13T15:36:26.513297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/hm-model-build-base/model_build_base.csv')#.sample(100000)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:36:26.516113Z","iopub.execute_input":"2022-11-13T15:36:26.516560Z","iopub.status.idle":"2022-11-13T15:38:47.919060Z","shell.execute_reply.started":"2022-11-13T15:36:26.516521Z","shell.execute_reply":"2022-11-13T15:38:47.915987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#filling in nulls for CUSTART predictors\ndf[['CUSTART_QUANTITY_SOLD_1M', 'CUSTART_QUANTITY_SOLD_3M', 'CUSTART_QUANTITY_SOLD_12M', 'CUSTART_QUANTITY_SOLD_OVERALL', 'CUSTART_NUM_CHANNEL_2', 'ART_AVERAGE_PRICE', 'ART_NUM_CHANNEL_2']]=df[['CUSTART_QUANTITY_SOLD_1M', 'CUSTART_QUANTITY_SOLD_3M', 'CUSTART_QUANTITY_SOLD_12M', 'CUSTART_QUANTITY_SOLD_OVERALL', 'CUSTART_NUM_CHANNEL_2', 'ART_AVERAGE_PRICE', 'ART_NUM_CHANNEL_2']].fillna(value=0)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:47.924394Z","iopub.execute_input":"2022-11-13T15:38:47.924908Z","iopub.status.idle":"2022-11-13T15:38:48.542905Z","shell.execute_reply.started":"2022-11-13T15:38:47.924859Z","shell.execute_reply":"2022-11-13T15:38:48.541515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df[df['ART_QUANTITY_SOLD_1M']>0]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:48.544743Z","iopub.execute_input":"2022-11-13T15:38:48.546023Z","iopub.status.idle":"2022-11-13T15:38:51.095505Z","shell.execute_reply.started":"2022-11-13T15:38:48.545912Z","shell.execute_reply":"2022-11-13T15:38:51.094404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:51.097036Z","iopub.execute_input":"2022-11-13T15:38:51.097398Z","iopub.status.idle":"2022-11-13T15:38:51.109235Z","shell.execute_reply.started":"2022-11-13T15:38:51.097364Z","shell.execute_reply":"2022-11-13T15:38:51.107810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:51.111483Z","iopub.execute_input":"2022-11-13T15:38:51.111911Z","iopub.status.idle":"2022-11-13T15:38:51.157271Z","shell.execute_reply.started":"2022-11-13T15:38:51.111877Z","shell.execute_reply":"2022-11-13T15:38:51.156359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mask=(df['SNAP_DATE']=='2020-09-08')|(df['SNAP_DATE']=='2020-09-01')|(df['SNAP_DATE']=='2020-09-15')\nvalid_mask=df['SNAP_DATE']=='2020-09-15'","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:51.161687Z","iopub.execute_input":"2022-11-13T15:38:51.162130Z","iopub.status.idle":"2022-11-13T15:38:52.164689Z","shell.execute_reply.started":"2022-11-13T15:38:51.162092Z","shell.execute_reply":"2022-11-13T15:38:52.163385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[train_mask]['SOLD'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:52.166694Z","iopub.execute_input":"2022-11-13T15:38:52.167314Z","iopub.status.idle":"2022-11-13T15:38:54.583206Z","shell.execute_reply.started":"2022-11-13T15:38:52.167255Z","shell.execute_reply":"2022-11-13T15:38:54.581932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[valid_mask]['SOLD'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:54.585394Z","iopub.execute_input":"2022-11-13T15:38:54.586311Z","iopub.status.idle":"2022-11-13T15:38:55.370278Z","shell.execute_reply.started":"2022-11-13T15:38:54.586258Z","shell.execute_reply":"2022-11-13T15:38:55.368732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration","metadata":{"execution":{"iopub.status.busy":"2022-11-06T11:00:13.387735Z","iopub.execute_input":"2022-11-06T11:00:13.388366Z","iopub.status.idle":"2022-11-06T11:00:13.395479Z","shell.execute_reply.started":"2022-11-06T11:00:13.388315Z","shell.execute_reply":"2022-11-06T11:00:13.393782Z"}}},{"cell_type":"code","source":"df[train_mask].describe()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:38:55.371861Z","iopub.execute_input":"2022-11-13T15:38:55.372263Z","iopub.status.idle":"2022-11-13T15:39:09.594622Z","shell.execute_reply.started":"2022-11-13T15:38:55.372227Z","shell.execute_reply":"2022-11-13T15:39:09.593414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[train_mask].describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:09.595931Z","iopub.execute_input":"2022-11-13T15:39:09.596331Z","iopub.status.idle":"2022-11-13T15:39:16.131683Z","shell.execute_reply.started":"2022-11-13T15:39:09.596296Z","shell.execute_reply":"2022-11-13T15:39:16.130491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data transformations","metadata":{}},{"cell_type":"code","source":"col_target='SOLD'\ncol_exclude=[\n'SNAP_DATE',\n'CUSTOMER_ID',\n'ARTICLE_ID',\n# 'ART_DAYS_SINCE_FIRST_PURCHASE',\n# 'ART_DAYS_SINCE_LAST_PURCHASE',\n\ncol_target,\n    \n\n]#+[col for col in df.columns if 'CUSTART' in col]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.133291Z","iopub.execute_input":"2022-11-13T15:39:16.134092Z","iopub.status.idle":"2022-11-13T15:39:16.140465Z","shell.execute_reply.started":"2022-11-13T15:39:16.134049Z","shell.execute_reply":"2022-11-13T15:39:16.138624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_exclude","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.142388Z","iopub.execute_input":"2022-11-13T15:39:16.142860Z","iopub.status.idle":"2022-11-13T15:39:16.159066Z","shell.execute_reply.started":"2022-11-13T15:39:16.142813Z","shell.execute_reply":"2022-11-13T15:39:16.157735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Categorical transformations","metadata":{}},{"cell_type":"code","source":"import category_encoders as ce","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.161239Z","iopub.execute_input":"2022-11-13T15:39:16.162297Z","iopub.status.idle":"2022-11-13T15:39:16.417395Z","shell.execute_reply.started":"2022-11-13T15:39:16.162246Z","shell.execute_reply":"2022-11-13T15:39:16.416325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# potential predictors without encoding\ncat_preds = [col for col in df.select_dtypes(include=\"object\") if col not in col_exclude]\nbool_preds = [col for col in df.select_dtypes(include=\"bool\") if col not in col_exclude]\ndatetime_preds = [col for col in df.select_dtypes(include=\"datetime\") if col not in col_exclude]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.418759Z","iopub.execute_input":"2022-11-13T15:39:16.419160Z","iopub.status.idle":"2022-11-13T15:39:16.811803Z","shell.execute_reply.started":"2022-11-13T15:39:16.419123Z","shell.execute_reply":"2022-11-13T15:39:16.810333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_preds","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.813343Z","iopub.execute_input":"2022-11-13T15:39:16.813972Z","iopub.status.idle":"2022-11-13T15:39:16.821843Z","shell.execute_reply.started":"2022-11-13T15:39:16.813930Z","shell.execute_reply":"2022-11-13T15:39:16.820481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Target Encoding","metadata":{"execution":{"iopub.status.busy":"2022-11-06T11:18:50.992838Z","iopub.execute_input":"2022-11-06T11:18:50.993263Z","iopub.status.idle":"2022-11-06T11:18:51.002318Z","shell.execute_reply.started":"2022-11-06T11:18:50.993229Z","shell.execute_reply":"2022-11-06T11:18:51.000747Z"}}},{"cell_type":"code","source":"# # bayesian target encoding\nencoder = ce.TargetEncoder(min_samples_leaf=1, smoothing=1.0)\nencoder.fit_transform(df[train_mask][cat_preds], df[train_mask][col_target])\n\ndf = pd.concat([df, encoder.transform(df[cat_preds]).add_prefix(\"BAYES_\")], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:16.823270Z","iopub.execute_input":"2022-11-13T15:39:16.823609Z","iopub.status.idle":"2022-11-13T15:39:55.467184Z","shell.execute_reply.started":"2022-11-13T15:39:16.823571Z","shell.execute_reply":"2022-11-13T15:39:55.465364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_preds=[col for col in df.select_dtypes(include=[\"int\",\"float\"]) if col not in col_exclude]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:55.469172Z","iopub.execute_input":"2022-11-13T15:39:55.470413Z","iopub.status.idle":"2022-11-13T15:39:58.862984Z","shell.execute_reply.started":"2022-11-13T15:39:55.470368Z","shell.execute_reply":"2022-11-13T15:39:58.861793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(col_preds)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:58.864870Z","iopub.execute_input":"2022-11-13T15:39:58.865272Z","iopub.status.idle":"2022-11-13T15:39:58.872775Z","shell.execute_reply.started":"2022-11-13T15:39:58.865239Z","shell.execute_reply":"2022-11-13T15:39:58.871398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fitting models","metadata":{}},{"cell_type":"code","source":"(df[train_mask][col_target]>0).value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:39:58.874633Z","iopub.execute_input":"2022-11-13T15:39:58.875311Z","iopub.status.idle":"2022-11-13T15:40:00.934312Z","shell.execute_reply.started":"2022-11-13T15:39:58.875268Z","shell.execute_reply":"2022-11-13T15:40:00.932839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGBoost model","metadata":{}},{"cell_type":"markdown","source":"## Fitting model","metadata":{}},{"cell_type":"code","source":"xgb_model = XGBClassifier(max_depth=4, seed=seed, colsample_bytree=0.5, gamma=1, min_child_weight=5, n_estimators=100)\nxgb_model.fit(df[train_mask].loc[:, col_preds], df[train_mask][col_target], verbose=0, eval_metric='logloss')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:40:00.936136Z","iopub.execute_input":"2022-11-13T15:40:00.936536Z","iopub.status.idle":"2022-11-13T15:47:43.945890Z","shell.execute_reply.started":"2022-11-13T15:40:00.936503Z","shell.execute_reply":"2022-11-13T15:47:43.944382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['XGB_SCORE']=xgb_model.predict_proba(df[col_preds])[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:47:43.947839Z","iopub.execute_input":"2022-11-13T15:47:43.948217Z","iopub.status.idle":"2022-11-13T15:47:50.991208Z","shell.execute_reply.started":"2022-11-13T15:47:43.948180Z","shell.execute_reply":"2022-11-13T15:47:50.989941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The Lift on the train set is: '+ str(mofr.metrics.lift(df[train_mask][col_target], df[train_mask]['XGB_SCORE'])))\nprint('The gini on the train set is: '+ str(mofr.metrics.gini(df[train_mask][col_target], df[train_mask]['XGB_SCORE'])))\nprint('The accuracy on the train set is: '+ str(mofr.metrics.accuracy_score(df[train_mask][col_target], df[train_mask]['XGB_SCORE'].apply(lambda x: int(x>0.5)))))\nprint('\\n')\nprint('The Lift on the valid set is: '+ str(mofr.metrics.lift(df[valid_mask][col_target], df[valid_mask]['XGB_SCORE'])))\nprint('The gini on the valid set is: '+ str(mofr.metrics.gini(df[valid_mask][col_target], df[valid_mask]['XGB_SCORE'])))\nprint('The accuracy on the valid set is: '+ str(mofr.metrics.accuracy_score(df[valid_mask][col_target], df[valid_mask]['XGB_SCORE'].apply(lambda x: int(x>0.5)))))\nprint('\\n')\n\nrandom_customers=pd.DataFrame(np.random.choice(df[valid_mask]['CUSTOMER_ID'].unique(), size=1000))\nrandom_customers=random_customers.rename(columns={0:'CUSTOMER_ID'})\ntop12=df[(valid_mask)].merge(random_customers)[['SOLD','CUSTOMER_ID', 'ARTICLE_ID', 'XGB_SCORE']].groupby('CUSTOMER_ID').apply(lambda x : x.sort_values(by = 'XGB_SCORE', ascending = False).head(12).reset_index(drop = True)).reset_index(drop = True)\nprecision=pd.DataFrame(top12.groupby('CUSTOMER_ID')['SOLD'].apply(np.mean)).rename(columns={'SOLD':'PRECISION'}).reset_index()\nmean_precision=np.mean(precision['PRECISION'])\nprint(f'The mean precision on valid set is: {mean_precision}')","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:47:50.992927Z","iopub.execute_input":"2022-11-13T15:47:50.993594Z","iopub.status.idle":"2022-11-13T15:48:17.767589Z","shell.execute_reply.started":"2022-11-13T15:47:50.993543Z","shell.execute_reply":"2022-11-13T15:48:17.766262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from mofr.basic_evaluators.ROCCurve import ROCCurveEvaluator\ndf['one']=1\n\nrce=ROCCurveEvaluator()\nrce.d(df[valid_mask]).t([(col_target,'one')]).s(['XGB_SCORE'])\nrce.get_graph()\n\ndel df['one']","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:17.769580Z","iopub.execute_input":"2022-11-13T15:48:17.770060Z","iopub.status.idle":"2022-11-13T15:48:20.983079Z","shell.execute_reply.started":"2022-11-13T15:48:17.770014Z","shell.execute_reply":"2022-11-13T15:48:20.982022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_idx = xgb_model.feature_importances_.argsort()\norder_ = []\nfor i in sorted_idx:\n  order_.append(col_preds[i])\nplt.figure(figsize=(10, 10))\nfig = plt.barh(order_, xgb_model.feature_importances_[sorted_idx])\nplt.xlabel(\"Xgboost Feature Importance\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:20.989442Z","iopub.execute_input":"2022-11-13T15:48:20.990571Z","iopub.status.idle":"2022-11-13T15:48:22.632066Z","shell.execute_reply.started":"2022-11-13T15:48:20.990514Z","shell.execute_reply":"2022-11-13T15:48:22.630655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# results=[]\n# for col in col_preds:\n#     results.append((col, np.abs(mofr.metrics.gini(df[valid_mask][col_target], df[valid_mask][col]))))\n  \n# pd.DataFrame(results, columns=['Predictor', 'GINI']).sort_values(by='GINI', ascending=False)[0:30]","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:22.633700Z","iopub.execute_input":"2022-11-13T15:48:22.634100Z","iopub.status.idle":"2022-11-13T15:48:22.639765Z","shell.execute_reply.started":"2022-11-13T15:48:22.634065Z","shell.execute_reply":"2022-11-13T15:48:22.638689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Saving the model","metadata":{}},{"cell_type":"code","source":"import pickle\nfile_name = \"hm_xgb_model.pkl\"\n\n# save\npickle.dump(xgb_model, open(file_name, \"wb\"))\n\n# # load\n# #xgb_model= pickle.load(open(file_name, \"rb\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:22.641409Z","iopub.execute_input":"2022-11-13T15:48:22.642350Z","iopub.status.idle":"2022-11-13T15:48:22.662305Z","shell.execute_reply.started":"2022-11-13T15:48:22.642308Z","shell.execute_reply":"2022-11-13T15:48:22.661057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfile_name = \"hm_encoder.pkl\"\n\n# save\npickle.dump(encoder, open(file_name, \"wb\"))\n\n# # load\n# #encoder = pickle.load(open(file_name, \"rb\"))","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:22.664233Z","iopub.execute_input":"2022-11-13T15:48:22.669666Z","iopub.status.idle":"2022-11-13T15:48:22.677373Z","shell.execute_reply.started":"2022-11-13T15:48:22.669610Z","shell.execute_reply":"2022-11-13T15:48:22.676129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## SHAP values","metadata":{}},{"cell_type":"code","source":"import shap  # package used to calculate Shap values","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:22.678868Z","iopub.execute_input":"2022-11-13T15:48:22.679288Z","iopub.status.idle":"2022-11-13T15:48:26.395289Z","shell.execute_reply.started":"2022-11-13T15:48:22.679251Z","shell.execute_reply":"2022-11-13T15:48:26.393876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"row_to_show = 1\ndata_for_prediction = df[train_mask][col_preds].iloc[row_to_show]  # use 1 row of data here. Could use multiple rows if desired\ndata_for_prediction_array = data_for_prediction.values.reshape(1, -1)\n\n# Create object that can calculate shap values\nexplainer = shap.TreeExplainer(xgb_model)\n\n# Calculate Shap values\nshap_values = explainer.shap_values(data_for_prediction_array)\nshap.initjs()\nshap.force_plot(explainer.expected_value, shap_values, data_for_prediction)","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:26.396770Z","iopub.execute_input":"2022-11-13T15:48:26.397532Z","iopub.status.idle":"2022-11-13T15:48:31.869859Z","shell.execute_reply.started":"2022-11-13T15:48:26.397483Z","shell.execute_reply":"2022-11-13T15:48:31.868692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap_values = explainer.shap_values(df[train_mask][col_preds])\nshap.summary_plot(shap_values, df[train_mask][col_preds])","metadata":{"execution":{"iopub.status.busy":"2022-11-13T15:48:31.871583Z","iopub.execute_input":"2022-11-13T15:48:31.872071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shap.dependence_plot('ART_AVERAGE_PRICE', shap_values, df[train_mask][col_preds], interaction_index=\"AGE\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Future predictions part","metadata":{}},{"cell_type":"code","source":"del df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nfile_name1= \"/kaggle/input/models/hm_xgb_model_custart_full.pkl\"\nfile_name2= \"/kaggle/input/models/hm_encoder_custart_full.pkl\"\n\n# load\nxgb_model= pickle.load(open(file_name1, \"rb\"))\nencoder= pickle.load(open(file_name2, \"rb\"))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles=pd.read_csv('/kaggle/input/hm-model-build-base/articles_predictions.csv')#.fillna(999)\ncustomers=pd.read_csv('/kaggle/input/hm-model-build-base/customers_prediction.csv').fillna(0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del customers['Unnamed: 0']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"customers['CUSTOMER_ID10']=customers['CUSTOMER_ID'].apply(lambda x: x[0:10])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"articles['ARTICLE_ID6']=articles['ARTICLE_ID'].apply(lambda x: int(str(x)[0:6]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scoring customers in batches to produce submission file\nFor each customer a simple logic was used to generate up to 500 candidate items.\nThe logic was using the following blend of products:\n* Top 100 products bought by the customer last month\n* Top 100 products bought by the customer overall\n* Top 100 products bought by all customers last month\n* Top 100 products bought by all customers last month\n* top 100 products for top PRODUCT_TYPE_NAME of the customer\n\nCombination of customers and suggested candidate items were stored in dataset 'model_suggested_items_enriched_sorted.csv'; this file is also enriched with the CUSTART predictors (I put only small sample of this dataset here);\n\nAt the end this approach was able to produce a solid score of ~0.0247 which is in the bronze medals. This was without using any special computational environment. The biggest difference made was by adding CUSTART predictors (on the level of customer and article), which pushed the performance from 0.007 area to close to bronze medals. \n\nIt was a pleasure solving this competition and amazing way to learn about recommender systems. Thanks to Kaggle community and H&M for this opportunity.","metadata":{"execution":{"iopub.status.busy":"2022-11-06T18:55:13.828054Z","iopub.execute_input":"2022-11-06T18:55:13.828620Z","iopub.status.idle":"2022-11-06T18:55:13.834468Z","shell.execute_reply.started":"2022-11-06T18:55:13.828574Z","shell.execute_reply":"2022-11-06T18:55:13.833557Z"}}},{"cell_type":"code","source":"submission=pd.DataFrame()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_chunk(chunk):\n    chunk=chunk.rename(columns={'CUSTOMER_ID':'CUSTOMER_ID10'}).drop_duplicates()\n    chunk.drop_duplicates(subset=['CUSTOMER_ID10', 'ARTICLE_ID'], inplace=True)\n    chunk=chunk.merge(articles, how='left', on='ARTICLE_ID')\n    chunk=chunk.merge(customers, how='left', left_on='CUSTOMER_ID10', right_on='CUSTOMER_ID10')\n    chunk = pd.concat([chunk, encoder.transform(chunk[cat_preds]).add_prefix(\"BAYES_\")], axis=1)\n    \n    chunk['XGB_SCORE']=xgb_model.predict_proba(chunk[xgb_model.feature_names_in_])[:, 1]\n    chunk['ARTICLE_ID']=chunk['ARTICLE_ID'].apply(str).apply(lambda x: x.zfill(10))\n    a=chunk[['CUSTOMER_ID', 'ARTICLE_ID', 'XGB_SCORE']].groupby('CUSTOMER_ID').apply(lambda x : x.sort_values(by = 'XGB_SCORE', ascending = False).head(12).reset_index(drop = True)).reset_index(drop = True)\n    b=pd.DataFrame(a.groupby('CUSTOMER_ID')['ARTICLE_ID'].apply(list).apply(' '.join)).reset_index(drop=False).rename(columns={'ARTICLE_ID':'PREDICTION'})\n    return b","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n=0\nchunksize = 10 ** 6\nwith pd.read_csv('/kaggle/input/suggested-items/model_suggested_items_enriched_sorted.csv', chunksize=chunksize) as reader:\n    for chunk in reader:\n        print(f'{n}: {round(n/3.94,2)} % done')\n        submission=pd.concat([submission,process_chunk(chunk)])\n        n+=1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.drop_duplicates(subset='CUSTOMER_ID',keep='first', inplace=True, ignore_index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}