{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-31T18:52:17.387566Z","iopub.execute_input":"2021-12-31T18:52:17.387975Z","iopub.status.idle":"2021-12-31T18:52:17.402893Z","shell.execute_reply.started":"2021-12-31T18:52:17.387934Z","shell.execute_reply":"2021-12-31T18:52:17.402156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/ad-tracking-fraud-detection/train.csv\").drop(columns=\"attributed_time\")\ntest = pd.read_csv(\"/kaggle/input/ad-tracking-fraud-detection/test.csv\")\n\nprint (\"length of train set is\", len(df))\nprint (\"length of test set is\", len(test), end='\\n\\n')\n\nprint (\"train set example\", df.head(3), sep='\\n', end='\\n')\nprint (\"test set example\", test.head(3), sep='\\n', end='\\n')","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:52:17.404162Z","iopub.execute_input":"2021-12-31T18:52:17.404364Z","iopub.status.idle":"2021-12-31T18:52:41.541667Z","shell.execute_reply.started":"2021-12-31T18:52:17.404338Z","shell.execute_reply":"2021-12-31T18:52:41.540826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First of all. Let's look on our target feature. Is there any class imbalance?","metadata":{}},{"cell_type":"code","source":"print (df.is_attributed.value_counts(), end='\\n\\n')\nprint (df.is_attributed.mean())","metadata":{"execution":{"iopub.status.busy":"2021-12-31T19:28:05.089558Z","iopub.execute_input":"2021-12-31T19:28:05.090971Z","iopub.status.idle":"2021-12-31T19:28:05.287662Z","shell.execute_reply.started":"2021-12-31T19:28:05.090902Z","shell.execute_reply":"2021-12-31T19:28:05.286611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As often happens, we have a strong class imbalance. We have to keep this in mind.","metadata":{}},{"cell_type":"code","source":"df.click_time = pd.to_datetime(df.click_time, infer_datetime_format=True)\ntest.click_time = pd.to_datetime(test.click_time, infer_datetime_format=True)\n\nprint (\"train set starts from\", df[\"click_time\"].min())\nprint (\"and ends\", df[\"click_time\"].max())\n\nprint (\"test set starts from\", test[\"click_time\"].min())\nprint (\"and ends\", test[\"click_time\"].max())","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:52:41.73591Z","iopub.execute_input":"2021-12-31T18:52:41.736149Z","iopub.status.idle":"2021-12-31T18:52:47.478783Z","shell.execute_reply.started":"2021-12-31T18:52:41.736118Z","shell.execute_reply":"2021-12-31T18:52:47.477893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we see, test set starts right after when train set is over. It is popular train\\test split for time series data","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\n\ndf.groupby(df.click_time.dt.hour).count().plot(kind=\"bar\", ax=ax)\nfigtext = \"Dataset distribution per hour\"\nlegend = \"samples\"\nax.legend(legend)\nplt.figtext(.4, -.03, figtext)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:52:47.480273Z","iopub.execute_input":"2021-12-31T18:52:47.480753Z","iopub.status.idle":"2021-12-31T18:52:51.450158Z","shell.execute_reply.started":"2021-12-31T18:52:47.480696Z","shell.execute_reply":"2021-12-31T18:52:51.449059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There is no 15th hour in training data. In coincidence our test set almost completely consists off 15th hour\n\nLet's go a little bit further and look at each day in train set separately.","metadata":{}},{"cell_type":"code","source":"for i in df.click_time.dt.day.unique():\n    fig, ax = plt.subplots()\n    legend = \"samples\"\n    figtext = \"Day number {}\".format(i)\n    _ = df[df.click_time.dt.day == i]\n    _.groupby(_.click_time.dt.hour).count().plot(kind=\"bar\", ax=ax)\n    ax.legend(legend)\n    plt.figtext(.4, -.03, figtext)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:52:51.451368Z","iopub.execute_input":"2021-12-31T18:52:51.452312Z","iopub.status.idle":"2021-12-31T18:53:07.458605Z","shell.execute_reply.started":"2021-12-31T18:52:51.452259Z","shell.execute_reply":"2021-12-31T18:53:07.457652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Things go even weirder. We have significant part of 6th day and some small parts of 7th and 9th days!\n\nAnd amount of samples per hour is completely different. We have to keep in mind this trait to when we will make our cross validation scheme.","metadata":{}},{"cell_type":"code","source":"_ = df.nunique().to_dict()\ncolumns, n_unique = _.keys(), _.values()\n\nfigtext = \"number of unique values per column\"\nplt.bar(columns, n_unique)\nplt.figtext(.4, -.03, figtext);","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:07.459664Z","iopub.execute_input":"2021-12-31T18:53:07.459919Z","iopub.status.idle":"2021-12-31T18:53:08.872359Z","shell.execute_reply.started":"2021-12-31T18:53:07.459891Z","shell.execute_reply":"2021-12-31T18:53:08.871418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"number of categories in app column is\", len(df.app.unique()))\n\ndf[\"app\"].value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:08.873437Z","iopub.execute_input":"2021-12-31T18:53:08.873646Z","iopub.status.idle":"2021-12-31T18:53:14.134653Z","shell.execute_reply.started":"2021-12-31T18:53:08.873612Z","shell.execute_reply":"2021-12-31T18:53:14.133697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"App distribution lookl like logarithmic one","metadata":{}},{"cell_type":"code","source":"df[df.app < 20].app.count() / len(df)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:14.13572Z","iopub.execute_input":"2021-12-31T18:53:14.135945Z","iopub.status.idle":"2021-12-31T18:53:15.128612Z","shell.execute_reply.started":"2021-12-31T18:53:14.135918Z","shell.execute_reply":"2021-12-31T18:53:15.12775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First 20 most popular apps cover almost 87% of all data","metadata":{}},{"cell_type":"code","source":"df.app.value_counts().apply(np.log).plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:15.131586Z","iopub.execute_input":"2021-12-31T18:53:15.132024Z","iopub.status.idle":"2021-12-31T18:53:20.744772Z","shell.execute_reply.started":"2021-12-31T18:53:15.131975Z","shell.execute_reply":"2021-12-31T18:53:20.7439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"number of categories in os column is\", len(df.os.unique()))\n\ndf.os.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:20.746304Z","iopub.execute_input":"2021-12-31T18:53:20.746982Z","iopub.status.idle":"2021-12-31T18:53:25.393529Z","shell.execute_reply.started":"2021-12-31T18:53:20.746941Z","shell.execute_reply":"2021-12-31T18:53:25.392705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.os < 20].os.count() / len(df)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:25.394733Z","iopub.execute_input":"2021-12-31T18:53:25.395Z","iopub.status.idle":"2021-12-31T18:53:26.903713Z","shell.execute_reply.started":"2021-12-31T18:53:25.394969Z","shell.execute_reply":"2021-12-31T18:53:26.902987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.os.value_counts().apply(np.log).plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:26.905145Z","iopub.execute_input":"2021-12-31T18:53:26.905684Z","iopub.status.idle":"2021-12-31T18:53:31.372306Z","shell.execute_reply.started":"2021-12-31T18:53:26.905639Z","shell.execute_reply":"2021-12-31T18:53:31.371488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print (\"Number of categories in os column is\", len(df.device.unique()))\n\n_ = df.device.value_counts()\n_.iloc[0] / len(df)","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:31.374036Z","iopub.execute_input":"2021-12-31T18:53:31.374353Z","iopub.status.idle":"2021-12-31T18:53:31.685847Z","shell.execute_reply.started":"2021-12-31T18:53:31.374312Z","shell.execute_reply":"2021-12-31T18:53:31.685101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"One most popular device accounts 94% of all samples","metadata":{}},{"cell_type":"code","source":"print (\"Number of categories in os column is\", len(df.channel.unique()))\n\ndf.channel.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:31.686939Z","iopub.execute_input":"2021-12-31T18:53:31.687163Z","iopub.status.idle":"2021-12-31T18:53:34.107124Z","shell.execute_reply.started":"2021-12-31T18:53:31.687135Z","shell.execute_reply":"2021-12-31T18:53:34.106218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.channel.value_counts().apply(np.log).plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2021-12-31T18:53:34.108454Z","iopub.execute_input":"2021-12-31T18:53:34.108687Z","iopub.status.idle":"2021-12-31T18:53:36.396358Z","shell.execute_reply.started":"2021-12-31T18:53:34.108657Z","shell.execute_reply":"2021-12-31T18:53:36.395785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look on ips and check the hypothesis that exist spam ips with very little conversion\n\nLet's remember that average conversion rate in training set is 0.0024","metadata":{}},{"cell_type":"code","source":"ips_pivot = pd.pivot_table(df, index=\"ip\", values=\"is_attributed\", aggfunc=[\"mean\", \"sum\", \"count\"])\n\nips_pivot.columns = ips_pivot.columns.to_series().str.join('_')\n\nips_pivot.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T19:21:05.198945Z","iopub.execute_input":"2021-12-31T19:21:05.199331Z","iopub.status.idle":"2021-12-31T19:21:08.803789Z","shell.execute_reply.started":"2021-12-31T19:21:05.199289Z","shell.execute_reply":"2021-12-31T19:21:08.802804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's ignore ips with less than 500 samples","metadata":{}},{"cell_type":"code","source":"ips_500 = ips_pivot[ips_pivot.count_is_attributed > 500]\n\nips_500.mean_is_attributed = ips_500.mean_is_attributed.round(decimals=4)\n\nips_500.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-31T19:40:05.951122Z","iopub.execute_input":"2021-12-31T19:40:05.951863Z","iopub.status.idle":"2021-12-31T19:40:05.966640Z","shell.execute_reply.started":"2021-12-31T19:40:05.951818Z","shell.execute_reply":"2021-12-31T19:40:05.965630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = ips_500.groupby(\"mean_is_attributed\").count()[\"count_is_attributed\"]\n_.plot(use_index=True, xlim=[0, 0.0024])\nplt.figtext(.2, -.03, \"number of channels depending on conversion rate\");","metadata":{"execution":{"iopub.status.busy":"2021-12-31T20:08:44.595685Z","iopub.execute_input":"2021-12-31T20:08:44.596021Z","iopub.status.idle":"2021-12-31T20:08:44.800468Z","shell.execute_reply.started":"2021-12-31T20:08:44.595987Z","shell.execute_reply":"2021-12-31T20:08:44.799809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = ips_500.groupby(\"mean_is_attributed\").sum()[\"count_is_attributed\"]\n_.plot(use_index=True, xlim=[0, 0.0024])\nplt.figtext(.2, -.03, \"number of samples depending on conversion rate\");","metadata":{"execution":{"iopub.status.busy":"2021-12-31T20:09:51.783462Z","iopub.execute_input":"2021-12-31T20:09:51.783853Z","iopub.status.idle":"2021-12-31T20:09:52.002640Z","shell.execute_reply.started":"2021-12-31T20:09:51.783813Z","shell.execute_reply":"2021-12-31T20:09:52.001768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we see. There is a lot of ips with about of zero conversion rate. Probably we should use this information for our future model","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}