{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":8540,"databundleVersionId":862041,"isSourceIdPinned":false},{"sourceType":"datasetVersion","sourceId":15045006,"datasetId":9631418,"databundleVersionId":15924543},{"sourceType":"datasetVersion","sourceId":15045000,"datasetId":9631412,"databundleVersionId":15924537}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:26:16.838212Z","iopub.execute_input":"2026-03-04T16:26:16.838484Z","iopub.status.idle":"2026-03-04T16:26:18.352913Z","shell.execute_reply.started":"2026-03-04T16:26:16.838457Z","shell.execute_reply":"2026-03-04T16:26:18.351978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport joblib\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    roc_auc_score,\n    average_precision_score,\n    confusion_matrix\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:27:46.849394Z","iopub.execute_input":"2026-03-04T16:27:46.849881Z","iopub.status.idle":"2026-03-04T16:27:47.886909Z","shell.execute_reply.started":"2026-03-04T16:27:46.849846Z","shell.execute_reply":"2026-03-04T16:27:47.885881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_row = 100_000_000\nn_rows = 2_000_000\n\ncols = [\n    'ip','app','device','os','channel','click_time','is_attributed'\n]\n\ndf = pd.read_csv(\n    \"/kaggle/input/competitions/talkingdata-adtracking-fraud-detection/train.csv\",\n    skiprows=range(1, start_row),\n    nrows=n_rows,\n    usecols=cols\n)\n\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:32:08.68992Z","iopub.execute_input":"2026-03-04T16:32:08.690985Z","iopub.status.idle":"2026-03-04T16:33:30.097922Z","shell.execute_reply.started":"2026-03-04T16:32:08.690932Z","shell.execute_reply":"2026-03-04T16:33:30.096947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['click_time'] = pd.to_datetime(df['click_time'])\n\ndf['day'] = df['click_time'].dt.day\ndf['hour'] = df['click_time'].dt.hour\ndf['day_of_week'] = df['click_time'].dt.dayofweek\ndf['day_of_year'] = df['click_time'].dt.dayofyear\ndf['month'] = df['click_time'].dt.month","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:33:33.64858Z","iopub.execute_input":"2026-03-04T16:33:33.649525Z","iopub.status.idle":"2026-03-04T16:33:34.240293Z","shell.execute_reply.started":"2026-03-04T16:33:33.649476Z","shell.execute_reply":"2026-03-04T16:33:34.239211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['hour_sin'] = np.sin(2 * np.pi * df['hour'] / 24)\ndf['hour_cos'] = np.cos(2 * np.pi * df['hour'] / 24)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:33:41.768623Z","iopub.execute_input":"2026-03-04T16:33:41.769713Z","iopub.status.idle":"2026-03-04T16:33:41.854916Z","shell.execute_reply.started":"2026-03-04T16:33:41.769674Z","shell.execute_reply":"2026-03-04T16:33:41.853923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.sort_values('click_time')\n\ndf['ip_time_diff'] = df.groupby('ip')['click_time'].diff().dt.total_seconds().fillna(0)\ndf['ip_app_time_diff'] = df.groupby(['ip','app'])['click_time'].diff().dt.total_seconds().fillna(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:33:48.568525Z","iopub.execute_input":"2026-03-04T16:33:48.568892Z","iopub.status.idle":"2026-03-04T16:33:49.495793Z","shell.execute_reply.started":"2026-03-04T16:33:48.568859Z","shell.execute_reply":"2026-03-04T16:33:49.494832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['ip_count'] = df.groupby('ip')['ip'].transform('count')\ndf['ip_app_count'] = df.groupby(['ip','app'])['app'].transform('count')\ndf['ip_device_os_count'] = df.groupby(['ip','device','os'])['ip'].transform('count')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:34:09.392965Z","iopub.execute_input":"2026-03-04T16:34:09.393773Z","iopub.status.idle":"2026-03-04T16:34:10.19402Z","shell.execute_reply.started":"2026-03-04T16:34:09.393726Z","shell.execute_reply":"2026-03-04T16:34:10.192899Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['unique_app_per_ip'] = df.groupby('ip')['app'].transform('nunique')\ndf['unique_channel_per_ip'] = df.groupby('ip')['channel'].transform('nunique')\ndf['unique_device_per_ip'] = df.groupby('ip')['device'].transform('nunique')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:35:03.969441Z","iopub.execute_input":"2026-03-04T16:35:03.969817Z","iopub.status.idle":"2026-03-04T16:35:04.614629Z","shell.execute_reply.started":"2026-03-04T16:35:03.969786Z","shell.execute_reply":"2026-03-04T16:35:04.613592Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['ip_hour_clicks'] = df.groupby(['ip','hour'])['ip'].transform('count')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:35:05.648629Z","iopub.execute_input":"2026-03-04T16:35:05.648968Z","iopub.status.idle":"2026-03-04T16:35:05.852138Z","shell.execute_reply.started":"2026-03-04T16:35:05.648937Z","shell.execute_reply":"2026-03-04T16:35:05.851167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['ip_day_hour'] = df.groupby(['ip','day','hour'])['ip'].transform('count')\ndf['ip_hour_channel'] = df.groupby(['ip','hour','channel'])['ip'].transform('count')\ndf['ip_hour_os'] = df.groupby(['ip','hour','os'])['ip'].transform('count')\ndf['ip_hour_app'] = df.groupby(['ip','hour','app'])['ip'].transform('count')\ndf['ip_hour_device'] = df.groupby(['ip','hour','device'])['ip'].transform('count')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:35:13.648392Z","iopub.execute_input":"2026-03-04T16:35:13.648758Z","iopub.status.idle":"2026-03-04T16:35:15.659538Z","shell.execute_reply.started":"2026-03-04T16:35:13.64872Z","shell.execute_reply":"2026-03-04T16:35:15.658479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['channel_per_ip'] = df.groupby('ip')['channel'].transform('nunique')\n\ndf['device_stability'] = (\n    df.groupby('ip')['device'].transform('nunique') / df['ip_count']\n)\n\ndf['app_stability'] = (\n    df.groupby('ip')['app'].transform('nunique') / df['ip_count']\n)\n\ndf['os_stability'] = (\n    df.groupby('ip')['os'].transform('nunique') / df['ip_count']\n)\n\ndf['click_intensity'] = df['ip_hour_clicks'] / df['ip_count']\n\ndf['ip_click_pressure'] = df['ip_count'] / (df['unique_device_per_ip'] + 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:35:20.853713Z","iopub.execute_input":"2026-03-04T16:35:20.854061Z","iopub.status.idle":"2026-03-04T16:35:21.702829Z","shell.execute_reply.started":"2026-03-04T16:35:20.854013Z","shell.execute_reply":"2026-03-04T16:35:21.70187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['app_channel'] = df['app'] * 1000 + df['channel']\ndf['device_os'] = df['device'] * 1000 + df['os']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:35:29.928527Z","iopub.execute_input":"2026-03-04T16:35:29.928854Z","iopub.status.idle":"2026-03-04T16:35:29.959446Z","shell.execute_reply.started":"2026-03-04T16:35:29.928829Z","shell.execute_reply":"2026-03-04T16:35:29.95812Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['app_te'] = df.groupby('app')['is_attributed'].transform('mean')\ndf['device_te'] = df.groupby('device')['is_attributed'].transform('mean')\ndf['os_te'] = df.groupby('os')['is_attributed'].transform('mean')\ndf['channel_te'] = df.groupby('channel')['is_attributed'].transform('mean')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:39:57.096141Z","iopub.execute_input":"2026-03-04T16:39:57.096554Z","iopub.status.idle":"2026-03-04T16:39:57.326685Z","shell.execute_reply.started":"2026-03-04T16:39:57.096515Z","shell.execute_reply":"2026-03-04T16:39:57.325825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"friend_features = [\n'app','device','os','channel','day_of_week','day_of_year','month',\n'ip_count','ip_day_hour','ip_hour_channel','ip_hour_os',\n'ip_hour_app','ip_hour_device','click_intensity','channel_per_ip',\n'device_stability','app_stability','os_stability','hour_sin',\n'hour_cos','app_channel','device_os','ip_click_pressure'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:40:14.4953Z","iopub.execute_input":"2026-03-04T16:40:14.495751Z","iopub.status.idle":"2026-03-04T16:40:14.501883Z","shell.execute_reply.started":"2026-03-04T16:40:14.495699Z","shell.execute_reply":"2026-03-04T16:40:14.500848Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"my_features = [\n'ip','app','device','os','channel',\n'day','hour','day_of_week',\n'ip_time_diff','ip_app_time_diff',\n'ip_count','ip_app_count',\n'ip_device_os_count',\n'unique_app_per_ip','unique_channel_per_ip','unique_device_per_ip',\n'ip_hour_clicks',\n'app_te','device_te','os_te','channel_te'\n]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:40:20.535594Z","iopub.execute_input":"2026-03-04T16:40:20.535932Z","iopub.status.idle":"2026-03-04T16:40:20.541371Z","shell.execute_reply.started":"2026-03-04T16:40:20.535902Z","shell.execute_reply":"2026-03-04T16:40:20.54029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = df['is_attributed']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:40:27.415308Z","iopub.execute_input":"2026-03-04T16:40:27.415654Z","iopub.status.idle":"2026-03-04T16:40:27.420494Z","shell.execute_reply.started":"2026-03-04T16:40:27.415625Z","shell.execute_reply":"2026-03-04T16:40:27.419407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"friend_model = joblib.load(\"/kaggle/input/datasets/syam574/jayarammodel/hybrid_1_XGB_LGB.pkl\")\nmy_model = joblib.load(\"/kaggle/input/datasets/syam574/syammodel/xgboost_hybrid_model.pkl\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:40:28.335262Z","iopub.execute_input":"2026-03-04T16:40:28.335588Z","iopub.status.idle":"2026-03-04T16:40:28.423865Z","shell.execute_reply.started":"2026-03-04T16:40:28.335559Z","shell.execute_reply":"2026-03-04T16:40:28.423123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_friend = friend_model.predict(df[friend_features])\npred_my = my_model.predict(df[my_features])\n\nproba_friend = friend_model.predict_proba(df[friend_features])[:,1]\nproba_my = my_model.predict_proba(df[my_features])[:,1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:40:29.620098Z","iopub.execute_input":"2026-03-04T16:40:29.620402Z","iopub.status.idle":"2026-03-04T16:42:20.161651Z","shell.execute_reply.started":"2026-03-04T16:40:29.620377Z","shell.execute_reply":"2026-03-04T16:42:20.160856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate(y_true, y_pred, y_prob):\n\n    print(\"Accuracy:\", accuracy_score(y_true,y_pred))\n    print(\"Precision:\", precision_score(y_true,y_pred))\n    print(\"Recall:\", recall_score(y_true,y_pred))\n    print(\"F1:\", f1_score(y_true,y_pred))\n    print(\"ROC AUC:\", roc_auc_score(y_true,y_prob))\n    print(\"PR AUC:\", average_precision_score(y_true,y_prob))\n    print(\"Confusion Matrix:\\n\", confusion_matrix(y_true,y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:42:23.475808Z","iopub.execute_input":"2026-03-04T16:42:23.476193Z","iopub.status.idle":"2026-03-04T16:42:23.483163Z","shell.execute_reply.started":"2026-03-04T16:42:23.476161Z","shell.execute_reply":"2026-03-04T16:42:23.482089Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"===== FRIEND MODEL =====\")\nevaluate(y, pred_friend, proba_friend)\n\nprint(\"\\n===== MY MODEL =====\")\nevaluate(y, pred_my, proba_my)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T16:42:26.815519Z","iopub.execute_input":"2026-03-04T16:42:26.815866Z","iopub.status.idle":"2026-03-04T16:42:29.885091Z","shell.execute_reply.started":"2026-03-04T16:42:26.815835Z","shell.execute_reply":"2026-03-04T16:42:29.883973Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}