{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Otto: EDA over sessions\n\n* this notebook provides basic visualizations of train & test sessions","metadata":{}},{"cell_type":"code","source":"!pip  install nb-black fastparquet","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-11-15T10:13:10.748471Z","iopub.execute_input":"2022-11-15T10:13:10.749113Z","iopub.status.idle":"2022-11-15T10:13:26.039584Z","shell.execute_reply.started":"2022-11-15T10:13:10.749073Z","shell.execute_reply":"2022-11-15T10:13:26.038429Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext lab_black\n\nfrom datetime import datetime\nimport json\nfrom multiprocessing import Pool\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom pandas.api.types import CategoricalDtype\n\n\nplt.style.use(\"ggplot\")\ndata_path = Path(\"../input/otto-recommender-system\")\ncache_path = Path(\"../input/otto-parquet\")\n\n\nclass Config:\n    chunk_size = 10_000\n    n_threds = 22\n    train = data_path / \"train.jsonl\"\n    test = data_path / \"test.jsonl\"\n    submission = data_path / \"sample_submission.csv\"\n    test_cache = cache_path / \"test.parquet\"\n    train_cache = cache_path / \"train.parquet\"\n    test_session_cache = cache_path / \"test_session.parquet\"\n    train_session_cache = cache_path / \"train_session.parquet\"","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:13:30.523139Z","iopub.execute_input":"2022-11-15T10:13:30.524199Z","iopub.status.idle":"2022-11-15T10:13:30.557296Z","shell.execute_reply.started":"2022-11-15T10:13:30.524154Z","shell.execute_reply":"2022-11-15T10:13:30.555849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import builtins\nimport types\n\n\ndef imports():\n    for name, val in globals().items():\n        # module imports\n        if isinstance(val, types.ModuleType):\n            yield name, val\n\n            # functions / callables\n        if hasattr(val, \"__call__\"):\n            yield name, val\n\n\n\"\"\"\nusage: If @noglobal decorator is specified, the function throws an exception if global variables are used in it.\n\nref: https://gist.github.com/raven38/4e4c3c7a179283c441f575d6e375510c\n\"\"\"\nnoglobal = lambda f: types.FunctionType(\n    f.__code__, dict(imports()), f.__name__, f.__defaults__, f.__closure__\n)","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:13:31.913741Z","iopub.execute_input":"2022-11-15T10:13:31.914127Z","iopub.status.idle":"2022-11-15T10:13:31.935206Z","shell.execute_reply.started":"2022-11-15T10:13:31.914095Z","shell.execute_reply":"2022-11-15T10:13:31.933403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@noglobal\ndef f(chunk):\n    df = chunk.explode(\"events\").reset_index(drop=True)\n    event_df = pd.json_normalize(df[\"events\"])\n    df = pd.concat([df, event_df], axis=1).drop(\"events\", axis=1)\n    return df\n\n\n@noglobal\ndef make_otto_dataframe(data_path, num_chunks):\n    chunks = pd.read_json(data_path, lines=True, chunksize=Config.chunk_size)\n    dfs = None\n    with Pool(Config.n_threds) as pool:\n        pbar = tqdm(pool.imap(f, chunks), total=num_chunks)\n        dfs = list(pbar)\n    out_df = pd.concat(dfs)\n    return out_df\n\n\n@noglobal\ndef preprocess_otto_dataframe(df):\n    df = df.sort_values([\"session\", \"ts\"])\n    category_type = CategoricalDtype([\"clicks\", \"carts\", \"orders\"])\n    df[\"type\"] = df[\"type\"].astype(category_type)\n    df[\"aid\"] = df[\"aid\"].astype(np.uint32)\n    df[\"session\"] = df[\"session\"].astype(np.uint32)\n    df[\"ts\"] = df[\"ts\"].astype(np.uint64)\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:13:32.145011Z","iopub.execute_input":"2022-11-15T10:13:32.145673Z","iopub.status.idle":"2022-11-15T10:13:32.184824Z","shell.execute_reply.started":"2022-11-15T10:13:32.145594Z","shell.execute_reply":"2022-11-15T10:13:32.182829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not Config.test_cache.exists():\n    print(\"Making dataframe...\")\n    test = make_otto_dataframe(Config.test, test_num_chunks)\n    print(\"Preprocess...\")\n    test = preprocess_otto_dataframe(test)\n    print(\"Making cache...\")\n    test.to_parquet(Config.test_cache, index=False, engine=\"fastparquet\")\n    print(\"Done\")\nelse:\n    test = pd.read_parquet(Config.test_cache, engine=\"fastparquet\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:13:32.348196Z","iopub.execute_input":"2022-11-15T10:13:32.348668Z","iopub.status.idle":"2022-11-15T10:13:34.371098Z","shell.execute_reply.started":"2022-11-15T10:13:32.348618Z","shell.execute_reply":"2022-11-15T10:13:34.369801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not Config.train_cache.exists():\n    print(\"Making dataframe...\")\n    train = make_otto_dataframe(Config.train, train_num_chunks)\n    print(\"Preprocess...\")\n    train = preprocess_otto_dataframe(train)\n    print(\"Making cache...\")\n    train.to_parquet(Config.train_cache, index=False, engine=\"fastparquet\")\n    print(\"Done\")\nelse:\n    train = pd.read_parquet(Config.train_cache, engine=\"fastparquet\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:13:34.373684Z","iopub.execute_input":"2022-11-15T10:13:34.374164Z","iopub.status.idle":"2022-11-15T10:13:55.621876Z","shell.execute_reply.started":"2022-11-15T10:13:34.374117Z","shell.execute_reply":"2022-11-15T10:13:55.620435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not Config.train_session_cache.exists():\n    train[\"ts\"] = pd.to_datetime(train[\"ts\"], unit=\"ms\")\n    df = pd.concat([train, pd.get_dummies(train[\"type\"])], axis=1)\n    train_session = df.groupby(\"session\").agg(\n        ts_min=(\"ts\", \"min\"),\n        ts_max=(\"ts\", \"max\"),\n        count=(\"ts\", \"count\"),\n        clicks=(\"clicks\", \"sum\"),\n        carts=(\"carts\", \"sum\"),\n        orders=(\"orders\", \"sum\"),\n        unique_aids=(\"aid\", \"nunique\"),\n    )\n    train_session[\"duration\"] = train_session[\"ts_max\"] - train_session[\"ts_min\"]\n    train_session.to_parquet(Config.train_session_cache, index=False, engine=\"fastparquet\")\nelse:\n    train_session = pd.read_parquet(Config.train_session_cache, engine=\"fastparquet\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:14:50.055780Z","iopub.execute_input":"2022-11-15T10:14:50.056172Z","iopub.status.idle":"2022-11-15T10:14:52.116812Z","shell.execute_reply.started":"2022-11-15T10:14:50.056140Z","shell.execute_reply":"2022-11-15T10:14:52.115093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not Config.test_session_cache.exists():\n    test[\"ts\"] = pd.to_datetime(test[\"ts\"], unit=\"ms\")\n    df = pd.concat([test, pd.get_dummies(test[\"type\"])], axis=1)\n    test_session = df.groupby(\"session\").agg(\n        ts_min=(\"ts\", \"min\"),\n        ts_max=(\"ts\", \"max\"),\n        count=(\"ts\", \"count\"),\n        clicks=(\"clicks\", \"sum\"),\n        carts=(\"carts\", \"sum\"),\n        orders=(\"orders\", \"sum\"),\n        unique_aids=(\"aid\", \"nunique\"),\n    )\n    test_session[\"duration\"] = test_session[\"ts_max\"] - test_session[\"ts_min\"]\n    test_session.to_parquet(Config.test_session_cache, index=False, engine=\"fastparquet\")\nelse:\n    test_session = pd.read_parquet(Config.test_session_cache, engine=\"fastparquet\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:14:57.227486Z","iopub.execute_input":"2022-11-15T10:14:57.227894Z","iopub.status.idle":"2022-11-15T10:14:58.149638Z","shell.execute_reply.started":"2022-11-15T10:14:57.227860Z","shell.execute_reply":"2022-11-15T10:14:58.148344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_session","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:14:58.151715Z","iopub.execute_input":"2022-11-15T10:14:58.152084Z","iopub.status.idle":"2022-11-15T10:14:58.188415Z","shell.execute_reply.started":"2022-11-15T10:14:58.152049Z","shell.execute_reply":"2022-11-15T10:14:58.187216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_session","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:14:58.193186Z","iopub.execute_input":"2022-11-15T10:14:58.194292Z","iopub.status.idle":"2022-11-15T10:14:58.215111Z","shell.execute_reply.started":"2022-11-15T10:14:58.194246Z","shell.execute_reply":"2022-11-15T10:14:58.213932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Rough visualization of train & test sessions","metadata":{}},{"cell_type":"code","source":"df1, df2 = train_session.copy(), test_session.copy()\nn_samples = 500\ndf1 = df1.sample(n_samples, random_state=123)\ndf2 = df2.sample(n_samples, random_state=456)\ntrain_test_term = (datetime(2022, 7, 31), datetime(2022, 9, 5))\n\n_, ax = plt.subplots(figsize=(12, 6))\n\nfor i, (x1, x2) in enumerate(zip(df1[\"ts_min\"].to_list(), df1[\"ts_max\"].to_list())):\n    ax.plot(\n        [x1, x2], [i, i], color=\"blue\", linestyle=\"solid\", linewidth=3, label=\"train\"\n    )\n\nfor i, (x1, x2) in enumerate(zip(df2[\"ts_min\"].to_list(), df2[\"ts_max\"].to_list())):\n    ax.plot([x1, x2], [i, i], color=\"red\", linestyle=\"solid\", linewidth=3, label=\"test\")\n\n\nax.set(\n    title=f\"Train & Test sessions (random {n_samples} samples)\", xlim=train_test_term\n)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:14:58.601449Z","iopub.execute_input":"2022-11-15T10:14:58.601867Z","iopub.status.idle":"2022-11-15T10:15:01.561082Z","shell.execute_reply.started":"2022-11-15T10:14:58.601831Z","shell.execute_reply":"2022-11-15T10:15:01.559296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Duration of a session","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\nax1.hist(train_session[\"duration\"].astype(\"timedelta64[D]\"))\nax1.set(title=\"train\", xlabel=\"duration [day]\", ylabel=\"count\")\nax2.hist(test_session[\"duration\"].astype(\"timedelta64[D]\"))\nax2.set(title=\"test\", xlabel=\"duration [day]\", ylabel=\"count\")\nplt.suptitle(\"Duration\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:01.563978Z","iopub.execute_input":"2022-11-15T10:15:01.564835Z","iopub.status.idle":"2022-11-15T10:15:02.488341Z","shell.execute_reply.started":"2022-11-15T10:15:01.564784Z","shell.execute_reply":"2022-11-15T10:15:02.486941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\nax1.hist(\n    train_session[\"duration\"].astype(\"timedelta64[D]\"),\n    cumulative=True,\n    density=1,\n    bins=100,\n)\nax1.set(title=\"train\", xlabel=\"duration [day]\", ylabel=\"density\")\nax2.hist(\n    test_session[\"duration\"].astype(\"timedelta64[D]\"),\n    cumulative=True,\n    density=1,\n    bins=100,\n)\nax2.set(title=\"test\", xlabel=\"duration [day]\", ylabel=\"density\")\nplt.suptitle(\"Duration (CDF)\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:02.489697Z","iopub.execute_input":"2022-11-15T10:15:02.490169Z","iopub.status.idle":"2022-11-15T10:15:03.770507Z","shell.execute_reply.started":"2022-11-15T10:15:02.490117Z","shell.execute_reply":"2022-11-15T10:15:03.769332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1, df2 = train_session.copy(), test_session.copy()\n\n_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\ng1 = ax1.hist(\n    df1[\"duration\"].astype(\"timedelta64[s]\") / (1 * 60 * 60),\n    cumulative=True,\n    density=1,\n    bins=2000,\n)\nax1.set(title=\"train\", xlabel=\"duration [hour]\", ylabel=\"density\", xlim=(0, 1))\ng2 = ax2.hist(\n    df2[\"duration\"].astype(\"timedelta64[s]\") / (1 * 60 * 60),\n    cumulative=True,\n    density=1,\n    bins=2000,\n)\nax2.set(title=\"test\", xlabel=\"duration [hour]\", ylabel=\"density\", xlim=(0, 1))\nplt.suptitle(f\"Duration (CDF; $\\leq 1 hour$)\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:03.773319Z","iopub.execute_input":"2022-11-15T10:15:03.773712Z","iopub.status.idle":"2022-11-15T10:15:12.276553Z","shell.execute_reply.started":"2022-11-15T10:15:03.773676Z","shell.execute_reply":"2022-11-15T10:15:12.275351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"densities, bins = g2[:2]\npd.DataFrame({\"bins[min]\": bins[1:] * 60, \"density\": densities}).head(5)","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:12.278043Z","iopub.execute_input":"2022-11-15T10:15:12.278434Z","iopub.status.idle":"2022-11-15T10:15:12.297274Z","shell.execute_reply.started":"2022-11-15T10:15:12.278400Z","shell.execute_reply":"2022-11-15T10:15:12.296068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* most of input test sessions are very short: about **72%** of test sessions has only $\\leq 5$ minutes duration.","metadata":{}},{"cell_type":"markdown","source":"# #events in a session","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\n\nax1.hist(train_session[\"count\"], bins=100, log=True)\nax1.set(title=\"train\", xlabel=\"#events\", ylabel=\"count\")\nax2.hist(test_session[\"count\"], bins=100, log=True)\nax2.set(title=\"test\", xlabel=\"#events\", ylabel=\"count\")\n\nplt.suptitle(\"#Events\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:12.299060Z","iopub.execute_input":"2022-11-15T10:15:12.300019Z","iopub.status.idle":"2022-11-15T10:15:14.106805Z","shell.execute_reply.started":"2022-11-15T10:15:12.299979Z","shell.execute_reply":"2022-11-15T10:15:14.104685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\n\nax1.hist(train_session[\"count\"], bins=100, cumulative=True, density=1)\nax1.set(title=\"train\", xlabel=\"#events\", ylabel=\"count\")\nax2.hist(test_session[\"count\"], bins=100, cumulative=True, density=1)\nax2.set(title=\"test\", xlabel=\"#events\", ylabel=\"count\")\n\nplt.suptitle(\"#Events (CDF)\")","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:14.109128Z","iopub.execute_input":"2022-11-15T10:15:14.109676Z","iopub.status.idle":"2022-11-15T10:15:15.190248Z","shell.execute_reply.started":"2022-11-15T10:15:14.109600Z","shell.execute_reply":"2022-11-15T10:15:15.189105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1, df2 = train_session.copy(), test_session.copy()\n\n_, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 3), sharex=True)\n\ng1 = ax1.hist(df1[\"count\"], bins=500, cumulative=True, density=1)\nax1.set(title=\"train\", xlabel=\"#events\", ylabel=\"count\", xlim=(0, 20))\ng2 = ax2.hist(df2[\"count\"], bins=500, cumulative=True, density=1)\nax2.set(title=\"test\", xlabel=\"#events\", ylabel=\"count\", xlim=(0, 20))\n\nplt.suptitle(f\"#Events (CDF; $\\leq 20$ events)\")\nprint(g2[0][:10])","metadata":{"execution":{"iopub.status.busy":"2022-11-15T10:15:15.192295Z","iopub.execute_input":"2022-11-15T10:15:15.192748Z","iopub.status.idle":"2022-11-15T10:15:18.315195Z","shell.execute_reply.started":"2022-11-15T10:15:15.192708Z","shell.execute_reply":"2022-11-15T10:15:18.313980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* quite a lot of events in test sessions are very few: about **45%** of test session contains only 1 event. That means, for 45% of data, we should predict future events by only one event.","metadata":{}}]}