{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":5174,"databundleVersionId":45685,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir('/kaggle/input/avito-duplicate-ads-detection')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:33:28.749137Z","iopub.execute_input":"2025-10-14T23:33:28.749634Z","iopub.status.idle":"2025-10-14T23:33:28.770699Z","shell.execute_reply.started":"2025-10-14T23:33:28.749606Z","shell.execute_reply":"2025-10-14T23:33:28.769695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Unzip just the files you need into the working directory\nwith zipfile.ZipFile('/kaggle/input/avito-duplicate-ads-detection/ItemPairs_train.csv.zip', 'r') as z:\n    z.extractall('/kaggle/working')\n\nwith zipfile.ZipFile('/kaggle/input/avito-duplicate-ads-detection/ItemInfo_train.csv.zip', 'r') as z:\n    z.extractall('/kaggle/working')\n\n# Load them\npairs = pd.read_csv('/kaggle/working/ItemPairs_train.csv')\ninfo = pd.read_csv('/kaggle/working/ItemInfo_train.csv')\n\nprint(\"Pairs shape:\", pairs.shape)\nprint(\"Info shape:\", info.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:33:45.255361Z","iopub.execute_input":"2025-10-14T23:33:45.255683Z","iopub.status.idle":"2025-10-14T23:35:28.854001Z","shell.execute_reply.started":"2025-10-14T23:33:45.255663Z","shell.execute_reply":"2025-10-14T23:35:28.852758Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pairs_sample = pairs.sample(100000, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:36:03.538306Z","iopub.execute_input":"2025-10-14T23:36:03.53863Z","iopub.status.idle":"2025-10-14T23:36:03.756387Z","shell.execute_reply.started":"2025-10-14T23:36:03.538606Z","shell.execute_reply":"2025-10-14T23:36:03.755081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile, os\n\n# List files to verify the path\nprint(os.listdir('/kaggle/input/avito-duplicate-ads-detection'))\n\n# Unzip the training pairs and info files\nwith zipfile.ZipFile('/kaggle/input/avito-duplicate-ads-detection/ItemPairs_train.csv.zip', 'r') as z:\n    z.extractall('/kaggle/working')\n\nwith zipfile.ZipFile('/kaggle/input/avito-duplicate-ads-detection/ItemInfo_train.csv.zip', 'r') as z:\n    z.extractall('/kaggle/working')\n\nprint(\"✅ Files unzipped successfully!\")\nprint(os.listdir('/kaggle/working'))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:38:29.747395Z","iopub.execute_input":"2025-10-14T23:38:29.747709Z","iopub.status.idle":"2025-10-14T23:38:58.838922Z","shell.execute_reply.started":"2025-10-14T23:38:29.747687Z","shell.execute_reply":"2025-10-14T23:38:58.837737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load only necessary columns and a small sample\npairs = pd.read_csv(\n    '/kaggle/working/ItemPairs_train.csv',\n    usecols=['itemID_1', 'itemID_2', 'isDuplicate'],\n    nrows=200000  # read only first 200k rows\n)\n\ninfo = pd.read_csv(\n    '/kaggle/working/ItemInfo_train.csv',\n    usecols=['itemID', 'price'],\n    nrows=200000\n)\n\nprint(\"Pairs shape:\", pairs.shape)\nprint(\"Info shape:\", info.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:42:25.821087Z","iopub.execute_input":"2025-10-14T23:42:25.821825Z","iopub.status.idle":"2025-10-14T23:42:29.313988Z","shell.execute_reply.started":"2025-10-14T23:42:25.821792Z","shell.execute_reply":"2025-10-14T23:42:29.312915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Read sample data (with only needed columns)\npairs = pd.read_csv('/kaggle/working/ItemPairs_train.csv', usecols=['itemID_1', 'itemID_2', 'isDuplicate'], nrows=200000)\ninfo = pd.read_csv('/kaggle/working/ItemInfo_train.csv', usecols=['itemID', 'price'], nrows=200000)\n\n# Merge step 1: info as i1\ninfo_1 = info.rename(columns={'itemID': 'itemID_1', 'price': 'price_1'})\nmerged = pairs.merge(info_1, on='itemID_1', how='left')\n\n# Merge step 2: info as i2\ninfo_2 = info.rename(columns={'itemID': 'itemID_2', 'price': 'price_2'})\nmerged = merged.merge(info_2, on='itemID_2', how='left')\n\n# Compute price difference\nmerged['price_diff'] = abs(merged['price_1'] - merged['price_2'])\nmerged = merged[merged['isDuplicate'] == 1]\n\n# Plot\nsns.histplot(data=merged, x='price_diff', bins=50, color='teal')\nplt.title('Price Difference Distribution for Duplicate Ads')\nplt.xlabel('Absolute Price Difference')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:45:24.248187Z","iopub.execute_input":"2025-10-14T23:45:24.248557Z","iopub.status.idle":"2025-10-14T23:45:26.74684Z","shell.execute_reply.started":"2025-10-14T23:45:24.248531Z","shell.execute_reply":"2025-10-14T23:45:26.745743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(merged[['itemID_1','itemID_2','price_1','price_2','price_diff']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:45:31.349825Z","iopub.execute_input":"2025-10-14T23:45:31.350213Z","iopub.status.idle":"2025-10-14T23:45:31.3706Z","shell.execute_reply.started":"2025-10-14T23:45:31.350191Z","shell.execute_reply":"2025-10-14T23:45:31.369325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Drop rows with missing prices\nmerged = merged.dropna(subset=['price_1', 'price_2'])\n\n# Recalculate price difference\nmerged['price_diff'] = abs(merged['price_1'] - merged['price_2'])\n\n# Check again\nprint(merged[['itemID_1','itemID_2','price_1','price_2','price_diff']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:46:38.108956Z","iopub.execute_input":"2025-10-14T23:46:38.109281Z","iopub.status.idle":"2025-10-14T23:46:38.123862Z","shell.execute_reply.started":"2025-10-14T23:46:38.109243Z","shell.execute_reply":"2025-10-14T23:46:38.122805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.histplot(data=merged, x='price_diff', bins=50, color='teal')\nplt.title('Price Difference Distribution for Duplicate Ads')\nplt.xlabel('Absolute Price Difference')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:46:59.261109Z","iopub.execute_input":"2025-10-14T23:46:59.261495Z","iopub.status.idle":"2025-10-14T23:46:59.514019Z","shell.execute_reply.started":"2025-10-14T23:46:59.261469Z","shell.execute_reply":"2025-10-14T23:46:59.51306Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport pandas as pd\n\npairs = pd.read_csv('/kaggle/working/ItemPairs_train.csv', usecols=['isDuplicate'], nrows=200000)\n\ndup_rate = pairs['isDuplicate'].mean() * 100\n\nsns.countplot(x='isDuplicate', data=pairs, palette='pastel')\nplt.title(f'Overall Duplicate vs Non-Duplicate Ads (Duplication Rate: {dup_rate:.2f}%)')\nplt.xlabel('Duplicate Flag (0 = Not Duplicate, 1 = Duplicate)')\nplt.ylabel('Number of Ad Pairs')\nplt.show()\n\nprint(f\"Overall duplication rate: {dup_rate:.2f}%\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:48:36.612136Z","iopub.execute_input":"2025-10-14T23:48:36.612752Z","iopub.status.idle":"2025-10-14T23:48:36.823637Z","shell.execute_reply.started":"2025-10-14T23:48:36.612724Z","shell.execute_reply":"2025-10-14T23:48:36.82256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(10,6))\nsns.barplot(data=cat_dup_df, x='Category ID', y='Duplicate Rate (%)', palette='viridis')\n\nplt.xticks(rotation=45)\nplt.title('Top 10 Categories by Duplicate Rate (%)')\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:59:33.000445Z","iopub.execute_input":"2025-10-14T23:59:33.000803Z","iopub.status.idle":"2025-10-14T23:59:33.267145Z","shell.execute_reply.started":"2025-10-14T23:59:33.000777Z","shell.execute_reply":"2025-10-14T23:59:33.265859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load sample subset\npairs = pd.read_csv('/kaggle/working/ItemPairs_train.csv', usecols=['isDuplicate'], nrows=50000)\n\n# Simulate a shorter date range (e.g., 2 months of hourly data)\npairs['ingestion_date'] = pd.date_range('2025-01-01', periods=len(pairs), freq='h').date\n\n# Aggregate by date\ntrend = pairs.groupby('ingestion_date', as_index=False)['isDuplicate'].mean()\n\n# Keep only the first 30 days for a shorter time window\ntrend_short = trend.head(30)\n\n# Plot area chart\nplt.figure(figsize=(10,5))\nplt.fill_between(trend_short['ingestion_date'], trend_short['isDuplicate']*100, color='orange', alpha=0.4)\nplt.plot(trend_short['ingestion_date'], trend_short['isDuplicate']*100, color='orange', linewidth=2)\nplt.title('Duplicate Rate Over First 30 Days (Area Chart)')\nplt.xlabel('Date')\nplt.ylabel('Duplicate Rate (%)')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T23:55:54.134338Z","iopub.execute_input":"2025-10-14T23:55:54.134679Z","iopub.status.idle":"2025-10-14T23:55:54.442854Z","shell.execute_reply.started":"2025-10-14T23:55:54.134657Z","shell.execute_reply":"2025-10-14T23:55:54.441695Z"}},"outputs":[],"execution_count":null}]}