{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30788,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-28T00:40:25.248287Z","iopub.execute_input":"2024-10-28T00:40:25.248673Z","iopub.status.idle":"2024-10-28T00:40:25.278823Z","shell.execute_reply.started":"2024-10-28T00:40:25.248641Z","shell.execute_reply":"2024-10-28T00:40:25.278047Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install --upgrade tensorflow\n!pip install pandas\n!pip install numpy\n!pip install lightgbm\n!pip install scikit-learn\n!pip install joblib\n!pip install polars","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:41:48.148492Z","iopub.execute_input":"2024-10-28T00:41:48.148895Z","iopub.status.idle":"2024-10-28T00:42:17.268769Z","shell.execute_reply.started":"2024-10-28T00:41:48.148864Z","shell.execute_reply":"2024-10-28T00:42:17.267659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport joblib\n\nfrom sklearn.model_selection import TimeSeriesSplit\nfrom sklearn.metrics import r2_score\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:42:17.270776Z","iopub.execute_input":"2024-10-28T00:42:17.271102Z","iopub.status.idle":"2024-10-28T00:42:17.409083Z","shell.execute_reply.started":"2024-10-28T00:42:17.271070Z","shell.execute_reply":"2024-10-28T00:42:17.408306Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check TensorFlow version\nprint(f\"TensorFlow version: {tf.__version__}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:42:17.410085Z","iopub.execute_input":"2024-10-28T00:42:17.410372Z","iopub.status.idle":"2024-10-28T00:42:17.414673Z","shell.execute_reply.started":"2024-10-28T00:42:17.410343Z","shell.execute_reply":"2024-10-28T00:42:17.414033Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Detect TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n    print('Not connected to a TPU')\n\n# Initialize TPU if available\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\n    print(\"TPU initialized.\")\nelse:\n    strategy = tf.distribute.get_strategy()  # Default strategy\n    print(\"Running on CPU/GPU.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:42:17.416043Z","iopub.execute_input":"2024-10-28T00:42:17.416291Z","iopub.status.idle":"2024-10-28T00:42:17.447950Z","shell.execute_reply.started":"2024-10-28T00:42:17.416267Z","shell.execute_reply":"2024-10-28T00:42:17.447315Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_dir = '/kaggle/input/jane-street-real-time-market-data-forecasting/'\n\n# List all files and directories\nfor root, dirs, files in os.walk(data_dir):\n    level = root.replace(data_dir, '').count(os.sep)\n    indent = ' ' * 4 * (level)\n    print(f\"{indent}{os.path.basename(root)}/\")\n    subindent = ' ' * 4 * (level + 1)\n    for f in files:\n        print(f\"{subindent}{f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:42:18.196498Z","iopub.execute_input":"2024-10-28T00:42:18.196855Z","iopub.status.idle":"2024-10-28T00:42:18.221651Z","shell.execute_reply.started":"2024-10-28T00:42:18.196827Z","shell.execute_reply":"2024-10-28T00:42:18.220894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize an empty list to store DataFrames\ntrain_dfs = []\n\n# Path to train.parquet\ntrain_dir = os.path.join(data_dir, 'train.parquet')\n\n# Iterate through all partition folders and load parquet files\nfor partition in os.listdir(train_dir):\n    partition_path = os.path.join(train_dir, partition)\n    if os.path.isdir(partition_path):\n        for file in os.listdir(partition_path):\n            if file.endswith('.parquet'):\n                file_path = os.path.join(partition_path, file)\n                print(f\"Loading {file_path}...\")\n                df = pl.read_parquet(file_path)\n                train_dfs.append(df)\n\n# Concatenate all partitions into a single DataFrame\ntrain = pl.concat(train_dfs, how='vertical')\nprint(f\"Total training rows: {train.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:44:02.667350Z","iopub.execute_input":"2024-10-28T00:44:02.667748Z","iopub.status.idle":"2024-10-28T00:44:08.195700Z","shell.execute_reply.started":"2024-10-28T00:44:02.667719Z","shell.execute_reply":"2024-10-28T00:44:08.194911Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load lags.parquet\nlags_path = os.path.join(data_dir, 'lags.parquet')\nlags = pl.read_parquet(lags_path)\nprint(f\"Lags data shape: {lags.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:44:08.888806Z","iopub.execute_input":"2024-10-28T00:44:08.889164Z","iopub.status.idle":"2024-10-28T00:44:08.899049Z","shell.execute_reply.started":"2024-10-28T00:44:08.889134Z","shell.execute_reply":"2024-10-28T00:44:08.898119Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load test.parquet\ntest_path = os.path.join(data_dir, 'test.parquet')\ntest = pl.read_parquet(test_path)\nprint(f\"Test data shape: {test.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:44:09.912995Z","iopub.execute_input":"2024-10-28T00:44:09.913743Z","iopub.status.idle":"2024-10-28T00:44:09.926595Z","shell.execute_reply.started":"2024-10-28T00:44:09.913702Z","shell.execute_reply":"2024-10-28T00:44:09.925383Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load features.csv\nfeatures_path = os.path.join(data_dir, 'features.csv')\nfeatures_df = pd.read_csv(features_path)\nprint(f\"Features metadata shape: {features_df.shape}\")\n\n# Load responders.csv\nresponders_path = os.path.join(data_dir, 'responders.csv')\nresponders_df = pd.read_csv(responders_path)\nprint(f\"Responders metadata shape: {responders_df.shape}\")\n\n# Load sample_submission.csv\nsample_submission_path = os.path.join(data_dir, 'sample_submission.csv')\nsample_submission_df = pd.read_csv(sample_submission_path)\nprint(f\"Sample submission shape: {sample_submission_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:44:12.326236Z","iopub.execute_input":"2024-10-28T00:44:12.327074Z","iopub.status.idle":"2024-10-28T00:44:12.342263Z","shell.execute_reply.started":"2024-10-28T00:44:12.327033Z","shell.execute_reply":"2024-10-28T00:44:12.341168Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert training data to Pandas for preprocessing\ntrain_pd = train.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:50:54.736532Z","iopub.execute_input":"2024-10-28T00:50:54.736951Z","iopub.status.idle":"2024-10-28T00:50:55.866804Z","shell.execute_reply.started":"2024-10-28T00:50:54.736919Z","shell.execute_reply":"2024-10-28T00:50:55.865938Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the number and percentage of missing values per feature\nmissing_counts = train_pd.isnull().sum()\nmissing_percent = (missing_counts / len(train_pd)) * 100\n\n# Combine counts and percentages into a single DataFrame for better readability\nmissing_summary = pd.DataFrame({\n    'Missing_Count': missing_counts,\n    'Missing_Percent': missing_percent\n})\n\n# Filter features with missing values\nmissing_summary = missing_summary[missing_summary['Missing_Count'] > 0]\n\n# Sort the summary by percentage of missingness in descending order\nmissing_summary = missing_summary.sort_values(by='Missing_Percent', ascending=False)\n\nprint(\"Missing values per feature:\")\nprint(missing_summary)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:51:12.652073Z","iopub.execute_input":"2024-10-28T00:51:12.652567Z","iopub.status.idle":"2024-10-28T00:51:17.753442Z","shell.execute_reply.started":"2024-10-28T00:51:12.652533Z","shell.execute_reply":"2024-10-28T00:51:17.752705Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming train_pd is your Pandas DataFrame after loading the data\n\n# Define the threshold for missingness\nmissing_threshold_percent = 20\n\n# Calculate the percentage of missing values per feature\nmissing_counts = train_pd.isnull().sum()\nmissing_percent = (missing_counts / len(train_pd)) * 100\n\n# Create a DataFrame summarizing missingness\nmissing_summary = pd.DataFrame({\n    'Missing_Count': missing_counts,\n    'Missing_Percent': missing_percent\n})\n\n# Filter features with missing percentage less than the threshold\nacceptable_features = missing_summary[missing_summary['Missing_Percent'] < missing_threshold_percent].index.tolist()\n\n# Exclude the target variable from features list if present\nif 'responder_6' in acceptable_features:\n    acceptable_features.remove('responder_6')\n\nprint(f\"Number of features with less than {missing_threshold_percent}% missingness: {len(acceptable_features)}\")\nprint(\"List of acceptable features:\")\nprint(acceptable_features)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:54:01.448260Z","iopub.execute_input":"2024-10-28T00:54:01.449078Z","iopub.status.idle":"2024-10-28T00:54:06.597296Z","shell.execute_reply.started":"2024-10-28T00:54:01.449041Z","shell.execute_reply":"2024-10-28T00:54:06.596452Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Define the target variable\ntarget = 'responder_6'\n\n# Compute Pearson correlation between each acceptable feature and the target\ncorrelations = train_pd[acceptable_features + [target]].corr()[target].abs().sort_values(ascending=False)\n\n# Exclude the target itself from the list\ncorrelations = correlations.drop(target)\n\n# Display the top 20 most correlated features\ntop_20_correlated = correlations.head(20)\nprint(\"Top 20 features correlated with responder_6:\")\nprint(top_20_correlated)","metadata":{"execution":{"iopub.status.busy":"2024-10-28T00:54:23.819856Z","iopub.execute_input":"2024-10-28T00:54:23.820220Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Define the target variable\ntarget = 'responder_6'\n\n# Compute Pearson correlation between each acceptable feature and the target\ncorrelations = train_pd[acceptable_features + [target]].corr()[target].abs().sort_values(ascending=False)\n\n# Exclude the target itself from the list\ncorrelations = correlations.drop(target)\n\n# Display the top 20 most correlated features\ntop_20_correlated = correlations.head(20)\nprint(\"Top 20 features correlated with responder_6:\")\nprint(top_20_correlated)","metadata":{},"outputs":[],"execution_count":null}]}