{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Introduction","metadata":{}},{"cell_type":"markdown","source":"**Recommendation Using Item to Item GNN Embedding**\n\n1. [Create dataset](https://www.kaggle.com/code/cafelatte1/otto-create-dataset-gnn-embedding)\n2. [Training](https://www.kaggle.com/cafelatte1/otto-training-gnn-embedding)\n3. [Inference](https://www.kaggle.com/cafelatte1/otto-inference-gnn-embedding)","metadata":{}},{"cell_type":"markdown","source":"## Setup","metadata":{}},{"cell_type":"code","source":"!nvcc -V","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:06:55.172100Z","iopub.execute_input":"2023-01-15T13:06:55.172575Z","iopub.status.idle":"2023-01-15T13:06:56.319799Z","shell.execute_reply.started":"2023-01-15T13:06:55.172479Z","shell.execute_reply":"2023-01-15T13:06:56.318311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # UPGRADE TORCH VERSION TO FIT WITH TORCH GEOMETRIC GPU\n# !pip install -q torch==1.12.1+cu102 --no-cache-dir --extra-index-url https://download.pytorch.org/whl/cu102","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:06:56.322248Z","iopub.execute_input":"2023-01-15T13:06:56.323443Z","iopub.status.idle":"2023-01-15T13:06:56.328781Z","shell.execute_reply.started":"2023-01-15T13:06:56.323393Z","shell.execute_reply":"2023-01-15T13:06:56.327578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GLOBAL_SEED = 42\n\nimport os\nos.environ[\"PYTHONIOENCODING\"] = \"utf8\"\nos.environ['PYTHONHASHSEED'] = str(GLOBAL_SEED)\nimport sys\nfrom glob import glob\n\nimport pandas as pd\nimport numpy as np\nfrom numpy import random as np_rnd\nimport random as rnd\nimport shutil\nimport gc\nimport datetime\nfrom collections import defaultdict, Counter\nfrom tqdm import tqdm\nfrom multiprocessing import Pool, cpu_count\nimport time\nimport pickle\n\nimport sklearn as skl\nfrom sklearn import model_selection\n\nimport torch\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch import nn\nimport torch.nn.functional as F\nfrom torch.optim import AdamW, Adam, SparseAdam\nfrom transformers import get_polynomial_decay_schedule_with_warmup\n\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\ntorch.__version__\n\nif torch.cuda.is_available():\n    import cudf\n    import cuml","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:06:56.330780Z","iopub.execute_input":"2023-01-15T13:06:56.331566Z","iopub.status.idle":"2023-01-15T13:07:03.706556Z","shell.execute_reply.started":"2023-01-15T13:06:56.331524Z","shell.execute_reply":"2023-01-15T13:07:03.705498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in glob(\"/kaggle/input/pytorch-geometric/pytorch_geometric_gpu/*\" if torch.cuda.is_available() else \"/kaggle/input/pytorch-geometric/pytorch_geometric_cpu/*\") :\n    !pip install -q --no-cache-dir {i}\n    \n!pip install -q --no-cache-dir torch-geometric\n\nfrom torch_geometric.data import Data\nfrom torch_geometric.utils import coalesce, is_undirected, to_undirected, sort_edge_index\nfrom torch_geometric.sampler import BaseSampler\nfrom torch_geometric.nn import GCNConv","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:07:03.709597Z","iopub.execute_input":"2023-01-15T13:07:03.710247Z","iopub.status.idle":"2023-01-15T13:08:22.734871Z","shell.execute_reply.started":"2023-01-15T13:07:03.710184Z","shell.execute_reply":"2023-01-15T13:08:22.733444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_everything(seed=42):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    # python random\n    rnd.seed(seed)\n    # numpy random\n    np_rnd.seed(seed)\n    # tf random\n    try:\n        tf_rnd.set_seed(seed)\n    except:\n        pass\n    # RAPIDS random\n    try:\n        cupy.random.seed(seed)\n    except:\n        pass\n    # pytorch random\n    try:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.backends.cudnn.deterministic = True\n    except:\n        pass\n\ndef pickleIO(obj, src, op=\"w\"):\n    if op == \"w\":\n        with open(src, op + \"b\") as f:\n            pickle.dump(obj, f)\n    elif op == \"r\":\n        with open(src, op + \"b\") as f:\n            tmp = pickle.load(f)\n        return tmp\n    else:\n        print(\"unknown operation\")\n        return obj\n    \ndef findIdx(data_x, col_names):\n    return [int(i) for i, j in enumerate(data_x) if j in col_names]\n\ndef createFolder(directory):\n    try:\n        if not os.path.exists(directory):\n            os.makedirs(directory)\n    except OSError:\n        print('Error: Creating directory. ' + directory)\n        \ndef create_submission(df):\n    df = df.reset_index()\n    df[\"type\"] = df[\"type\"].map(CFG.contentType_mapper)\n    df[\"session_type\"] = df[\"session\"].astype(\"str\") + \"_\" + df[\"type\"].astype(\"str\") + \"s\"\n    df = df[[\"session_type\", \"prediction\"]].rename({\"prediction\": \"labels\"}, axis=1)\n    return df\n\ndef create_get_ts(ts):\n    return int((ts.replace(tzinfo=CFG.tz) - CFG.ts_zero).total_seconds())\n\ndef visualize_graph(G, color):\n    plt.figure(figsize=(7,7))\n    plt.xticks([])\n    plt.yticks([])\n    nx.draw_networkx(G, pos=nx.spring_layout(G, seed=42), with_labels=False, node_color=color, cmap=\"Set2\")\n    plt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-01-15T13:08:22.738022Z","iopub.execute_input":"2023-01-15T13:08:22.738596Z","iopub.status.idle":"2023-01-15T13:08:22.754456Z","shell.execute_reply.started":"2023-01-15T13:08:22.738539Z","shell.execute_reply":"2023-01-15T13:08:22.753279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    local = False\n    debug = False\n    tz = datetime.timezone.utc\n    ts_zero = datetime.datetime(1970, 1, 1, tzinfo=tz)\n    contentType_mapper = pd.Series([\"clicks\", \"carts\", \"orders\"], index=[0, 1, 2])\n    target_weight = (0.1, 0.3, 0.6)\n    \n    n_folds = 5\n    batch_size = 1024\n    epochs = 50\n    early_stopping_rounds = 10\n    eta = 5e-4\n    weight_decay = 1e-4\n    max_grad_norm = 1e+2\n    embed_dim = 32\n    \nif CFG.local:\n    CFG.folder_path = \"./dataset/\"\nelse:\n    CFG.folder_path = \"/kaggle/input/\"\n\nif CFG.debug:\n    CFG.batch_size = 1024 * 64\nelse:\n    CFG.batch_size = 1024 * 256","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:08:22.755963Z","iopub.execute_input":"2023-01-15T13:08:22.756667Z","iopub.status.idle":"2023-01-15T13:08:22.784621Z","shell.execute_reply.started":"2023-01-15T13:08:22.756620Z","shell.execute_reply":"2023-01-15T13:08:22.783500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Data","metadata":{}},{"cell_type":"markdown","source":"**Graph Types**\n* Directed/Undirected graphs: Directed graphs are the ones that all the edges have directtions while in undirected graphs, the edges are does not have directions (Edge 방향성 유무)\n* Weighted/Binary graphs: Weighted graphs is a type of graph that each of the edges are assigned with a value while binary graphs are the ones that the edges does not have an assigned value. (Edge 가중치 존재 유무)\n* Homogenous/Heterogenous graphs: Homogenous graphs are the ones that all the nodes and/or edges are of the same type (e.g. friendship graph) while heterogenous graphs are graphs where the nodes and/or edges are of different types (e.g. knowledge graph). (노드 간 동질성 유무, 동질성이 있는 노드 그룹이 있어서 서로 묶지 못하면 Heterogenous)","metadata":{}},{"cell_type":"markdown","source":"**Attrs of Pytorch Data Class**\n\n* data.x: Node feature matrix with shape [num_nodes, num_node_features]\n\n* data.edge_index: Graph connectivity in COO format with shape [2, num_edges] and type torch.long\n\n* data.edge_attr: Edge feature matrix with shape [num_edges, num_edge_features]\n\n* data.y: Target to train against (may have arbitrary shape), e.g., node-level targets of shape [num_nodes, *] or graph-level targets of shape [1, *]\n\n* data.pos: Node position matrix with shape [num_nodes, num_dimensions]","metadata":{}},{"cell_type":"code","source":"fraction_of_sessions_to_use = 0.01 if CFG.debug else 0.5\n\ntrain = pd.read_parquet('../input/otto-full-optimized-memory-footprint/train.parquet')\ntest = pd.read_parquet('../input/otto-full-optimized-memory-footprint/test.parquet')\n    \nif fraction_of_sessions_to_use != 1:\n    lucky_sessions_train = train.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n    subset_of_train = train[train.session.isin(lucky_sessions_train)]\n    \n#     lucky_sessions_test = test.drop_duplicates(['session']).sample(frac=fraction_of_sessions_to_use, random_state=42)['session']\n#     subset_of_test = test[test.session.isin(lucky_sessions_test)]\n    subset_of_test = test\nelse:\n    subset_of_train = train\n    subset_of_test = test\n\nsubset_of_train.index = pd.MultiIndex.from_frame(subset_of_train[['session']])\nsubset_of_test.index = pd.MultiIndex.from_frame(subset_of_test[['session']])\n\n# Concat train & test data\n# This is not information leakage because of considering the predctions session by session\nsubsets = pd.concat([subset_of_train, subset_of_test])\n# subsets = subset_of_train\nsessions = subsets.session.unique()\n\ndel lucky_sessions_train, subset_of_train, subset_of_test; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:08:22.786541Z","iopub.execute_input":"2023-01-15T13:08:22.786920Z","iopub.status.idle":"2023-01-15T13:09:12.538331Z","shell.execute_reply.started":"2023-01-15T13:08:22.786884Z","shell.execute_reply":"2023-01-15T13:09:12.537166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Node feature & Edge information","metadata":{}},{"cell_type":"code","source":"chunk_size = 16384 * 2\nrecent_aids = 20\nhour_interval_threshold = 6\ndf_adj = pd.DataFrame(columns=[\"aid_x\", \"aid_y\"], dtype=\"int64\")\n\n# session 유니크 값 만큼 반복\nfor i in tqdm(range(0, sessions.shape[0], chunk_size), total=len(range(0, sessions.shape[0], chunk_size))):\n    # chunk size 만큼 session 선택 후 임시 데이터프레임 생성\n    current_chunk = subsets.loc[sessions[i]:sessions[min(sessions.shape[0]-1, i+chunk_size-1)]].reset_index(drop=True)\n    # 세션별로 가장 최근 N개 aid을 선택 (event 타입 상관 안 함)\n    current_chunk = current_chunk.groupby('session', as_index=False).nth(list(range(-recent_aids, 0))).reset_index(drop=True)\n    # 세션별 M * M 아이템 interaction row를 생성하는 operation\n    consecutive_AIDs = current_chunk.merge(current_chunk, on='session')\n    # 자기 자신과의 아이템 매칭은 제외\n    consecutive_AIDs = consecutive_AIDs[consecutive_AIDs.aid_x != consecutive_AIDs.aid_y]\n    # 두 아이템간의 interaction 시간 텀 계산\n    consecutive_AIDs['days_elapsed'] = (consecutive_AIDs.ts_y - consecutive_AIDs.ts_x) / 3600\n    # 두 아이템간 interaction 시간 텀이 threshold 값 이하인 아이템만 선택\n    consecutive_AIDs = consecutive_AIDs[(consecutive_AIDs.days_elapsed > 0) & (consecutive_AIDs.days_elapsed < hour_interval_threshold)]\n    # Create Edge Information\n    df_adj = pd.concat([df_adj, consecutive_AIDs[[\"aid_x\", \"aid_y\"]]], axis=0, ignore_index=True)\n    \nn_aids = subsets[\"aid\"].max() + 1\nnodes = np.arange(subsets[\"aid\"].max() + 1, dtype=\"int64\")","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:09:12.539642Z","iopub.execute_input":"2023-01-15T13:09:12.539995Z","iopub.status.idle":"2023-01-15T13:16:29.338228Z","shell.execute_reply.started":"2023-01-15T13:09:12.539963Z","shell.execute_reply":"2023-01-15T13:16:29.336476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Create Data pipeline with Pytorch Graph Data Class**","metadata":{}},{"cell_type":"code","source":"# Create Graph Data Class\ndata_graph = Data(\n    x=torch.tensor(nodes, dtype=torch.int64),\n    edge_index=torch.tensor(df_adj.to_numpy().T, dtype=torch.int64),\n)\n# Sort & Remove duplicates Edge indexes\ndata_graph.edge_index = coalesce(data_graph.edge_index)\n# Transform undirected graph\ndata_graph.edge_index = to_undirected(data_graph.edge_index)\npickleIO(data_graph, \"./data_graph.pkl\", \"w\")\n\ndel subsets, sessions, nodes, df_adj, current_chunk, consecutive_AIDs; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:16:29.340698Z","iopub.execute_input":"2023-01-15T13:16:29.341090Z","iopub.status.idle":"2023-01-15T13:17:54.924582Z","shell.execute_reply.started":"2023-01-15T13:16:29.341060Z","shell.execute_reply":"2023-01-15T13:17:54.923788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pickleIO(data_graph.x, \"./node_feature.pkl\", \"w\")","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:17:54.927484Z","iopub.execute_input":"2023-01-15T13:17:54.927838Z","iopub.status.idle":"2023-01-15T13:17:54.968368Z","shell.execute_reply.started":"2023-01-15T13:17:54.927806Z","shell.execute_reply":"2023-01-15T13:17:54.967380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"edge_index = pd.DataFrame(data_graph.edge_index.detach().cpu().numpy().T, columns=[\"x\", \"y\"], dtype=\"int64\")\nseed_everything()\nshuffled_idx = np_rnd.permutation(len(edge_index))\nedge_index.iloc[shuffled_idx[:int(len(shuffled_idx) * 0.8)]].reset_index(drop=True).to_parquet(\"./train_edge.parquet\")\nedge_index.iloc[shuffled_idx[int(len(shuffled_idx) * 0.8):]].reset_index(drop=True).to_parquet(\"./valid_edge.parquet\")","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:17:54.969832Z","iopub.execute_input":"2023-01-15T13:17:54.970256Z","iopub.status.idle":"2023-01-15T13:18:27.563645Z","shell.execute_reply.started":"2023-01-15T13:17:54.970209Z","shell.execute_reply":"2023-01-15T13:18:27.562160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"edge_index.iloc[shuffled_idx[:int(len(shuffled_idx) * 0.8)]].shape","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:18:27.565315Z","iopub.execute_input":"2023-01-15T13:18:27.566504Z","iopub.status.idle":"2023-01-15T13:18:38.125365Z","shell.execute_reply.started":"2023-01-15T13:18:27.566455Z","shell.execute_reply":"2023-01-15T13:18:38.124215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"edge_index.iloc[shuffled_idx[int(len(shuffled_idx) * 0.8):]].shape","metadata":{"execution":{"iopub.status.busy":"2023-01-15T13:18:38.126948Z","iopub.execute_input":"2023-01-15T13:18:38.127614Z","iopub.status.idle":"2023-01-15T13:18:40.753406Z","shell.execute_reply.started":"2023-01-15T13:18:38.127578Z","shell.execute_reply":"2023-01-15T13:18:40.752259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}