{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 1: Load and Explore the Data","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport kaggle_evaluation.jane_street_inference_server\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:36:48.333207Z","iopub.execute_input":"2024-10-21T21:36:48.333722Z","iopub.status.idle":"2024-10-21T21:36:48.366305Z","shell.execute_reply.started":"2024-10-21T21:36:48.333679Z","shell.execute_reply":"2024-10-21T21:36:48.364835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_responders = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\ndf_sample = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')\ndf_features = pd.read_csv('/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:36:50.344583Z","iopub.execute_input":"2024-10-21T21:36:50.345632Z","iopub.status.idle":"2024-10-21T21:36:50.363377Z","shell.execute_reply.started":"2024-10-21T21:36:50.345577Z","shell.execute_reply":"2024-10-21T21:36:50.362024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_responders.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:36:52.882267Z","iopub.execute_input":"2024-10-21T21:36:52.882757Z","iopub.status.idle":"2024-10-21T21:36:52.908571Z","shell.execute_reply.started":"2024-10-21T21:36:52.882681Z","shell.execute_reply":"2024-10-21T21:36:52.907230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sample.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:37:28.803080Z","iopub.execute_input":"2024-10-21T21:37:28.803513Z","iopub.status.idle":"2024-10-21T21:37:28.832923Z","shell.execute_reply.started":"2024-10-21T21:37:28.803471Z","shell.execute_reply":"2024-10-21T21:37:28.831425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display the first few rows to check the data\ndf_features.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T20:12:23.755093Z","iopub.execute_input":"2024-10-21T20:12:23.755537Z","iopub.status.idle":"2024-10-21T20:12:23.780086Z","shell.execute_reply.started":"2024-10-21T20:12:23.755494Z","shell.execute_reply":"2024-10-21T20:12:23.778508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:38:38.012527Z","iopub.execute_input":"2024-10-21T21:38:38.013074Z","iopub.status.idle":"2024-10-21T21:38:38.029307Z","shell.execute_reply.started":"2024-10-21T21:38:38.013027Z","shell.execute_reply":"2024-10-21T21:38:38.027521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\ndf_features.isnull().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T20:13:17.359545Z","iopub.execute_input":"2024-10-21T20:13:17.360734Z","iopub.status.idle":"2024-10-21T20:13:17.371864Z","shell.execute_reply.started":"2024-10-21T20:13:17.360678Z","shell.execute_reply":"2024-10-21T20:13:17.370133Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_features.describe().transpose()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:39:44.562939Z","iopub.execute_input":"2024-10-21T21:39:44.563532Z","iopub.status.idle":"2024-10-21T21:39:44.605215Z","shell.execute_reply.started":"2024-10-21T21:39:44.563473Z","shell.execute_reply":"2024-10-21T21:39:44.603617Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":" ## 2: Exploratory Data Analysis (EDA)","metadata":{}},{"cell_type":"markdown","source":"### A/ Distribution of True/False Values for Each Tag","metadata":{}},{"cell_type":"code","source":"# Get summary statistics for the binary tags (True/False counts)\ntag_columns = df_features.columns[1:]  # All tag columns\ntag_summary = df_features[tag_columns].apply(pd.Series.value_counts)\n\n# Plot the True/False distribution for each tag\nplt.figure(figsize=(12, 6))\ntag_summary.T.plot(kind='bar', stacked=True, color=['lightblue', 'lightcoral'])\nplt.title('Distribution of True/False Values for Each Tag')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show();\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T16:39:16.904575Z","iopub.execute_input":"2024-10-21T16:39:16.904893Z","iopub.status.idle":"2024-10-21T16:39:17.539414Z","shell.execute_reply.started":"2024-10-21T16:39:16.904859Z","shell.execute_reply":"2024-10-21T16:39:17.538352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.kdeplot(df_features)\nplt.show();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T21:09:10.061230Z","iopub.execute_input":"2024-10-21T21:09:10.061768Z","iopub.status.idle":"2024-10-21T21:09:10.627302Z","shell.execute_reply.started":"2024-10-21T21:09:10.061706Z","shell.execute_reply":"2024-10-21T21:09:10.626009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## 3: Data Preparation\n### Prepare the data for modeling in Scikit-learn. We will use tag_16 as the target variable.","metadata":{}},{"cell_type":"code","source":"# Check the data types of all columns\ndf_features.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T16:39:17.540626Z","iopub.execute_input":"2024-10-21T16:39:17.540952Z","iopub.status.idle":"2024-10-21T16:39:17.549532Z","shell.execute_reply.started":"2024-10-21T16:39:17.540917Z","shell.execute_reply":"2024-10-21T16:39:17.548411Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Identify non-numeric columns\nnon_numeric_columns = df_features.select_dtypes(include=['object']).columns\nprint(\"Non-numeric columns:\", non_numeric_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T16:39:17.550953Z","iopub.execute_input":"2024-10-21T16:39:17.551284Z","iopub.status.idle":"2024-10-21T16:39:17.560825Z","shell.execute_reply.started":"2024-10-21T16:39:17.551251Z","shell.execute_reply":"2024-10-21T16:39:17.559846Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Convert Non-Numeric Data","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\n\n# Apply Label Encoding to non-numeric columns\nlabel_encoder = LabelEncoder()\n\nfor col in non_numeric_columns:\n    df_features[col] = label_encoder.fit_transform(df_features[col])\n\n# Verify all columns are numeric now\ndf_features.dtypes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T16:39:17.561909Z","iopub.execute_input":"2024-10-21T16:39:17.562302Z","iopub.status.idle":"2024-10-21T16:39:17.646092Z","shell.execute_reply.started":"2024-10-21T16:39:17.562267Z","shell.execute_reply":"2024-10-21T16:39:17.644923Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### B/ Let's visualize the correlations between different features using a heatmap","metadata":{}},{"cell_type":"code","source":"# Calculate the correlation matrix for the binary tags\ncorr_matrix = df_features[tag_columns].astype(int).corr()\n\n# Plot a heatmap to visualize the correlations\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', vmin=-1, vmax=1)\nplt.title('Correlation Heatmap of Binary Tags')\nplt.tight_layout()\nplt.show();","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-21T16:39:17.647564Z","iopub.execute_input":"2024-10-21T16:39:17.648665Z","iopub.status.idle":"2024-10-21T16:39:18.830359Z","shell.execute_reply.started":"2024-10-21T16:39:17.648615Z","shell.execute_reply":"2024-10-21T16:39:18.829219Z"}},"outputs":[],"execution_count":null}]}