{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":96164,"databundleVersionId":11418275,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA) of the DRW Crypto Market Prediction Dataset \n\nThis analysis provides a detailed EDA focusing on data quality, variable relationships, and distributional characteristics using a sample of the dataset to avoid memory overload.","metadata":{"_uuid":"25b395be-9133-4ef0-9304-77decffff52f","_cell_guid":"eb698029-0996-428a-8882-384b81baa7c0","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import sys  \nimport random  \nimport numpy as np  \nimport pandas as pd  \nfrom pathlib import Path  \n\n# Version check for reproducibility  \nassert sys.version_info >= (3, 8), \"Python>=3.8 required\"  \nprint(f\"Python {sys.version_info.major}.{sys.version_info.minor}\")  \n\n# Seed for reproducibility  \nSEED = 42  \nrandom.seed(SEED)  \nnp.random.seed(SEED)  \n\n# File paths  \ndata_dir = Path('/kaggle/input/drw-crypto-market-prediction')  \ntrain_path = data_dir / 'train.parquet'  \ntest_path = data_dir / 'test.parquet'  \n\n# Load a sample of the dataset to save memory  \ndf_full = pd.read_parquet(train_path)  \ndf = df_full.sample(n=5000, random_state=SEED).reset_index(drop=True)  # Sample 5000 rows  \ndf_test = pd.read_parquet(test_path)  \nprint(f\"Original Train shape: {df_full.shape}, Sampled Train shape: {df.shape}, Test shape: {df_test.shape}\")  \n\n# Convert float64 columns to float32 for memory efficiency  \nfloat_cols = df.select_dtypes(include=['float64']).columns  \ndf[float_cols] = df[float_cols].astype('float32')  \n\ndf.head(3)","metadata":{"_uuid":"9b9b0455-8060-485b-956c-30e5c49501ac","_cell_guid":"34877d5f-0f60-427e-8f3b-5290d45dd912","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Basic data quality check  \ndef check_df(d):  \n    print(\"-- Shape --\", d.shape)  \n    print(\"-- Missing values --\", d.isna().sum().sum())  \n    infs = np.isinf(d.select_dtypes('number')).sum().sum()  \n    print(\"-- Infinities --\", infs)  \n    consts = [c for c in d if d[c].nunique() == 1]  \n    print(f\"-- Constant features ({len(consts)}) --\", consts[:5])  \n  \ncheck_df(df)","metadata":{"_uuid":"6d8c3dc6-a267-4edf-89d2-2fba0478c942","_cell_guid":"d63dafb7-ada1-46cd-9f91-948896cd192c","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 2. Data Validation and Cleaning","metadata":{"_uuid":"6a5056c0-5b24-41fd-bf6c-88d8b9c005cb","_cell_guid":"b882434d-6666-420b-a7c6-8e5de75238bc","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Handle missing values  \nfor col in df.columns:  \n    if df[col].dtype == 'object':  \n        df[col].fillna(df[col].mode()[0], inplace=True)  \n    else:  \n        df[col].fillna(df[col].median(), inplace=True)  \n  \n# Handle infinite values  \ndf.replace([np.inf, -np.inf], np.nan, inplace=True)  \ndf.fillna(df.max() * 1.1, inplace=True)  \n  \nprint(\"Missing values after cleaning:\", df.isna().sum().sum())  \nprint(\"Infinite values after cleaning:\", np.isinf(df.select_dtypes('number')).sum().sum())","metadata":{"_uuid":"e081ce60-f232-4434-83ac-419200c6e225","_cell_guid":"22747700-dde5-4128-ae14-f6b9b617645b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3. Univariate Non-Graphical EDA","metadata":{"_uuid":"b81da818-8ffc-4ca4-baf4-0f30fe793c80","_cell_guid":"17573716-b48c-4315-9d96-cf72d60192d6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"cat_cols = df.select_dtypes(include=['object']).columns  \nfor col in cat_cols:  \n    print(f\"Frequency table for {col}:\")  \n    print(df[col].value_counts(normalize=True))  \n    print()","metadata":{"_uuid":"b23d1f2f-1911-4a76-bba1-84366804dfd4","_cell_guid":"6e791563-1dff-4bad-bd7c-ab2343723e78","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"quant_cols = df.select_dtypes(include=['number']).columns  \nsummary_stats = df[quant_cols].describe().T  \nsummary_stats['skew'] = df[quant_cols].skew()  \nsummary_stats['kurtosis'] = df[quant_cols].kurtosis()  \nsummary_stats","metadata":{"_uuid":"7274723e-e5c0-43fa-88f7-362896c133c5","_cell_guid":"859bf434-7002-49fe-aece-090def58163f","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 4. Univariate Graphical EDA","metadata":{"_uuid":"8f8077ac-f56f-478b-8b59-c514be62090d","_cell_guid":"8304d276-d44b-497b-8765-010be69f54c0","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"import matplotlib.pyplot as plt  \nimport scipy.stats as st  \n\n# Histograms (limit to first 6 numeric columns to save memory)  \nfor col in quant_cols[:6]:  \n    plt.figure(figsize=(6, 4))  \n    plt.hist(df[col], bins=20, edgecolor='black')  \n    plt.title(f'Histogram of {col}')  \n    plt.xlabel(col)  \n    plt.ylabel('Frequency')  \n    plt.show()","metadata":{"_uuid":"e59503fe-8d6b-4ed6-a5b9-198cc9fd8a8b","_cell_guid":"0574f9ef-4134-40a0-a596-7705f18dfe44","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Boxplots (limit to first 6 numeric columns)  \nfor col in quant_cols[:6]:  \n    plt.figure(figsize=(6, 4))  \n    plt.boxplot(df[col].values, vert=False)  \n    plt.title(f'Boxplot of {col}')  \n    plt.show()","metadata":{"_uuid":"b334dbbc-0360-4a27-bf61-b7941af6d494","_cell_guid":"37e30061-fd8c-4aaf-b67d-9d937e342c72","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# QQ-plots (limit to first 6 numeric columns)  \nfor col in quant_cols[:6]:  \n    plt.figure(figsize=(6, 4))  \n    st.probplot(df[col], dist=\"norm\", plot=plt)  \n    plt.title(f'QQ-plot of {col}')  \n    plt.show()","metadata":{"_uuid":"5762ec4b-39ed-4771-a5cd-fb1c337e4cab","_cell_guid":"7b8840a0-164d-4d4d-b618-ff9df76fee1e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 5. Bivariate and Multivariate Non-Graphical EDA","metadata":{"_uuid":"24d0dde9-1c60-4051-ac53-05b9ab323e08","_cell_guid":"9a6442ba-31a4-423b-95b9-9f20c1a240dc","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Categorical-Categorical relationships  \nfor col1 in cat_cols:  \n    for col2 in cat_cols:  \n        if col1 != col2:  \n            print(f\"Cross-tabulation for {col1} and {col2}:\")  \n            print(pd.crosstab(df[col1], df[col2], normalize='index'))  \n            print()","metadata":{"_uuid":"8307133e-7868-4e3e-b57f-710e0d1f031b","_cell_guid":"25175153-21c0-431f-983c-e8258d87620a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Categorical-Quantitative pairs  \nfor cat_col in cat_cols:  \n    for quant_col in quant_cols[:6]:  # limit to first 6 for memory  \n        print(f\"Group summary for {cat_col} and {quant_col}:\")  \n        print(df.groupby(cat_col)[quant_col].agg(['mean', 'std']))  \n        print()","metadata":{"_uuid":"060adb46-4e02-4025-93ea-da17f2d88cfc","_cell_guid":"452ce019-b6a1-4946-b98d-bff446fd0606","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Correlation matrix (only numeric)  \ncorr_matrix = df[quant_cols].corr()  \nprint(\"Correlation matrix:\")  \nprint(corr_matrix)  \n\n# Strong correlations  \nstrong_corrs = corr_matrix[(corr_matrix > 0.7) | (corr_matrix < -0.7)]  \nprint(\"Strong correlations (|r| > 0.7):\")  \nprint(strong_corrs)","metadata":{"_uuid":"bb6f2adc-e398-4bb0-a444-28ec8a79d73c","_cell_guid":"d1ee2a8a-5d3c-4ab1-a24d-a7faf4b44914","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 6. Bivariate and Multivariate Graphical EDA","metadata":{"_uuid":"15ee5e12-1247-40b4-b0c4-6a9fe4933b85","_cell_guid":"8ca766a8-32bb-4c99-89d1-fb801d92896b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Scatterplots (limit to first 5 numeric columns)  \nfor i, col1 in enumerate(quant_cols[:5]):  \n    for col2 in quant_cols[:5]:  \n        if col1 != col2:  \n            plt.figure(figsize=(6, 4))  \n            plt.scatter(df[col1], df[col2], s=4, alpha=0.3)  \n            plt.title(f'{col1} vs {col2}')  \n            plt.xlabel(col1)  \n            plt.ylabel(col2)  \n            plt.show()","metadata":{"_uuid":"63c38877-71f3-4639-beec-5c02187591ef","_cell_guid":"7117d5ea-7934-49c0-b0d1-c10d1145a86e","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Side-by-side boxplots (limit to first 5 numeric columns)  \nfor cat_col in cat_cols:  \n    for quant_col in quant_cols[:5]:  \n        plt.figure(figsize=(6, 4))  \n        df.boxplot(column=quant_col, by=cat_col)  \n        plt.title(f'{quant_col} by {cat_col}')  \n        plt.suptitle('')  \n        plt.show()","metadata":{"_uuid":"edc6af1b-668a-4d1e-b441-2705ca90b10b","_cell_guid":"11a88117-ef57-4b3e-887d-ab0fc3ee43e7","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Pairplot (very limited columns to save memory)  \nimport seaborn as sns  \nsns.pairplot(df[quant_cols[:4]], diag_kind='kde')  \nplt.show()","metadata":{"_uuid":"3cc42e3c-ef8f-4bab-8790-96a98c8cd52e","_cell_guid":"a86a4fe2-69b9-4c32-9fc0-d4ae61cad03a","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 7. Feature Engineering Insights","metadata":{"_uuid":"e05cfa51-5141-4590-bd87-61e8385d554b","_cell_guid":"805e4c29-01fd-41f4-81e0-3375cb8dbd15","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Define imbalance (assuming these columns exist)  \ndf['imbalance'] = (df['bid_qty'] - df['ask_qty']) / (df['bid_qty'] + df['ask_qty'] + 1e-6)\n\n# Histogram of imbalance  \nplt.figure(figsize=(6, 4))  \nplt.hist(df['imbalance'], bins=100, edgecolor='black')  \nplt.title('Order Book Imbalance')  \nplt.xlabel('Imbalance')  \nplt.ylabel('Frequency')  \nplt.show()\n\n# Scatter with target  \nplt.figure(figsize=(6, 4))  \nplt.scatter(df['imbalance'], df['label'], s=4, alpha=0.3)  \nplt.title('Imbalance vs Label')  \nplt.xlabel('Imbalance')  \nplt.ylabel('Label')  \nplt.show()","metadata":{"_uuid":"ca92f1c9-9dff-49fd-b572-c37862ca6b6a","_cell_guid":"1f3394e4-a6f4-4904-ae29-963e98043be9","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 8. Data Quality and Robustness Checks","metadata":{"_uuid":"ef5c86bd-051b-40f6-b334-bf139544151d","_cell_guid":"4d08b1ac-2220-40c0-872c-6bf3e2f3e688","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"# Outlier removal (using 99th percentile)  \ndf_no_outliers = df[(df['bid_qty'] < df['bid_qty'].quantile(0.99)) & (df['ask_qty'] < df['ask_qty'].quantile(0.99))]  \nsummary_stats_no_outliers = df_no_outliers[quant_cols].describe().T  \nsummary_stats_no_outliers","metadata":{"_uuid":"195f5d58-5bb8-4043-97a5-0f6962f28e98","_cell_guid":"b2c5b554-c1b4-482d-892e-0b1e39d3abfd","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Missingness visualization  \nmissingness = df.isnull().sum()  \nplt.figure(figsize=(8, 4))  \nmissingness.plot(kind='bar')  \nplt.title('Missing Values by Feature')  \nplt.ylabel('Count')  \nplt.show()","metadata":{"_uuid":"b82c5e4e-daf0-4503-bedc-83d21fdb7377","_cell_guid":"a3916039-4d8c-4881-8b7a-d163499c5c71","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**This initial version represents a introductory foundational exploration, recognizing that there is still substantial work ahead as part of the continued pursuit, etc.**","metadata":{}}]}