{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport sklearn\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-12-28T06:33:32.103063Z","iopub.execute_input":"2022-12-28T06:33:32.103523Z","iopub.status.idle":"2022-12-28T06:33:32.124909Z","shell.execute_reply.started":"2022-12-28T06:33:32.103483Z","shell.execute_reply":"2022-12-28T06:33:32.123995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cudf\ntrain = cudf.read_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test.parquet')\ntrain_labels = cudf.read_parquet('/kaggle/input/otto-train-and-test-data-for-local-validation/test_labels.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:32.126565Z","iopub.execute_input":"2022-12-28T06:33:32.127445Z","iopub.status.idle":"2022-12-28T06:33:32.690548Z","shell.execute_reply.started":"2022-12-28T06:33:32.127407Z","shell.execute_reply":"2022-12-28T06:33:32.689497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_session_length(df):\n    # If not using cuDF, remove .to_pandas()\n    df['session_length'] = df.to_pandas().groupby('session')['ts'].transform('count')\n    return df\n\ndef add_action_num_reverse_chrono(df):\n    df['action_num_reverse_chrono'] = df.session_length - df.groupby('session').cumcount() - 1\n    return df\n\ndef add_log_recency_score(df):\n    linear_interpolation = 0.1 + ((1-0.1) / (df['session_length']-1)) * (df['session_length']-df['action_num_reverse_chrono']-1)\n    df['log_recency_score'] = (2 ** linear_interpolation - 1).fillna(1.0)\n    return df\n\ndef add_type_weighted_log_recency_score(df):\n    type_weights = {0:1, 1:6, 2:3}\n    df['type_weighted_log_recency_score'] = df['log_recency_score'] / df['type'].map(type_weights)\n    return df\n    \ndef apply(df, pipeline):\n    for f in pipeline:\n        df = f(df)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:32.691934Z","iopub.execute_input":"2022-12-28T06:33:32.692905Z","iopub.status.idle":"2022-12-28T06:33:32.702606Z","shell.execute_reply.started":"2022-12-28T06:33:32.692841Z","shell.execute_reply":"2022-12-28T06:33:32.701359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pipeline = [add_session_length, add_action_num_reverse_chrono, add_log_recency_score, add_type_weighted_log_recency_score]\n\ntrain = apply(train, pipeline)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:32.705659Z","iopub.execute_input":"2022-12-28T06:33:32.706406Z","iopub.status.idle":"2022-12-28T06:33:33.545400Z","shell.execute_reply.started":"2022-12-28T06:33:32.706354Z","shell.execute_reply":"2022-12-28T06:33:33.544175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ntype2id = {\"clicks\": 0, \"carts\": 1, \"orders\": 2}\n\ntrain_labels = train_labels.explode('ground_truth')\ntrain_labels = train_labels.rename(columns={'ground_truth': 'aid'})\ntrain_labels['type'] = train_labels.type.map(type2id)\n\ntrain_labels['aid'] = train_labels.aid.astype('int32')\ntrain_labels['type'] = train_labels.type.astype('uint8')\ntrain_labels['session'] = train_labels.session.astype('int32')\n\ntrain_labels['gt'] = 1\n\ntrain = train.merge(train_labels, on=['session', 'type', 'aid'], how='left')\ntrain['gt'] = train.gt.fillna(0)\ntrain = train.sort_values('session').reset_index(drop=True)\n\ntrain.head()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:33.547329Z","iopub.execute_input":"2022-12-28T06:33:33.548040Z","iopub.status.idle":"2022-12-28T06:33:33.743535Z","shell.execute_reply.started":"2022-12-28T06:33:33.548001Z","shell.execute_reply":"2022-12-28T06:33:33.742383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_session_lengths(df): # Fixed function name typo in original notebook\n    # In radek's version, each execution of this function returns a different list\n    # I don't think that is the intended behavior, and is fixed here\n    return df.groupby('session')['session_length'].count().sort_index().to_pandas().values\n\nsession_lengths_train = get_session_lengths(train)\n","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:33.747980Z","iopub.execute_input":"2022-12-28T06:33:33.750962Z","iopub.status.idle":"2022-12-28T06:33:33.794782Z","shell.execute_reply.started":"2022-12-28T06:33:33.750920Z","shell.execute_reply":"2022-12-28T06:33:33.793548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nimport lightgbm as lgb\nscoring = sklearn.metrics.make_scorer(sklearn.metrics.ndcg_score, greater_is_better=True)\nranker = model = lgb.LGBMRanker(\n    objective=\"lambdarank\",\n    metric=\"ndcg\",\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:33.797637Z","iopub.execute_input":"2022-12-28T06:33:33.798536Z","iopub.status.idle":"2022-12-28T06:33:34.864614Z","shell.execute_reply.started":"2022-12-28T06:33:33.798491Z","shell.execute_reply":"2022-12-28T06:33:34.863630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_cols = ['aid', 'type', 'action_num_reverse_chrono', 'session_length', 'log_recency_score', 'type_weighted_log_recency_score']\ntarget = 'gt'","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:34.865981Z","iopub.execute_input":"2022-12-28T06:33:34.867588Z","iopub.status.idle":"2022-12-28T06:33:34.872767Z","shell.execute_reply.started":"2022-12-28T06:33:34.867548Z","shell.execute_reply":"2022-12-28T06:33:34.871565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ranker = ranker.fit(\n    train[feature_cols].to_pandas(),\n    train[target].to_pandas(),\n    group=session_lengths_train,\n    callbacks=[lgb.reset_parameter(learning_rate=lambda x: max(0.01, 0.1 - 0.01 * x))]\n)","metadata":{"execution":{"iopub.status.busy":"2022-12-28T06:33:34.876704Z","iopub.execute_input":"2022-12-28T06:33:34.877061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = cudf.read_parquet('/kaggle/input/otto-full-optimized-memory-footprint/test.parquet')\ntest = apply(test, pipeline)\n\nscores = ranker.predict(test[feature_cols].to_pandas())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['score'] = scores\ntest_predictions = test.sort_values(by=['session', 'score'], \n                                    ascending=False)[['session', 'aid']] \\\n                                    .reset_index(drop=True)\n\ntest_predictions = test_predictions.to_pandas().groupby('session').head(20) \\\n                                   .groupby('session').agg(list) \\\n                                   .reset_index(drop=False) \n\nsession_types = []\nlabels = []\n\nfor session, preds in zip(test_predictions['session'].to_numpy(), test_predictions['aid'].to_numpy()):\n    l = ' '.join(str(p) for p in preds)\n    for session_type in ['clicks', 'carts', 'orders']:\n        labels.append(l)\n        session_types.append(f'{session}_{session_type}')\n\nsubmission = cudf.DataFrame({'session_type': session_types, 'labels': labels})\nsubmission.to_csv('submission.csv', index=False)\n\nsubmission.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}