{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-10-17T17:00:38.858920Z","iopub.execute_input":"2024-10-17T17:00:38.861351Z","iopub.status.idle":"2024-10-17T17:00:39.540854Z","shell.execute_reply.started":"2024-10-17T17:00:38.861259Z","shell.execute_reply":"2024-10-17T17:00:39.539553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom lightgbm import LGBMRegressor\nfrom sklearn.metrics import mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:39.544341Z","iopub.execute_input":"2024-10-17T17:00:39.545151Z","iopub.status.idle":"2024-10-17T17:00:42.097137Z","shell.execute_reply.started":"2024-10-17T17:00:39.545073Z","shell.execute_reply":"2024-10-17T17:00:42.095527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/features.csv')\nresponders_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/responders.csv')\nsample_data = pd.read_csv(r'/kaggle/input/jane-street-real-time-market-data-forecasting/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:42.098843Z","iopub.execute_input":"2024-10-17T17:00:42.099668Z","iopub.status.idle":"2024-10-17T17:00:42.127539Z","shell.execute_reply.started":"2024-10-17T17:00:42.099612Z","shell.execute_reply":"2024-10-17T17:00:42.126173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"features_data :\", features_data.shape)\nprint(\"responders_data :\", responders_data.shape)\nprint(\"sample_data :\", sample_data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:42.129568Z","iopub.execute_input":"2024-10-17T17:00:42.130094Z","iopub.status.idle":"2024-10-17T17:00:42.136742Z","shell.execute_reply.started":"2024-10-17T17:00:42.130038Z","shell.execute_reply":"2024-10-17T17:00:42.135570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:42.140457Z","iopub.execute_input":"2024-10-17T17:00:42.141266Z","iopub.status.idle":"2024-10-17T17:00:42.175509Z","shell.execute_reply.started":"2024-10-17T17:00:42.141213Z","shell.execute_reply":"2024-10-17T17:00:42.174073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:42.177078Z","iopub.execute_input":"2024-10-17T17:00:42.177548Z","iopub.status.idle":"2024-10-17T17:00:42.190837Z","shell.execute_reply.started":"2024-10-17T17:00:42.177501Z","shell.execute_reply":"2024-10-17T17:00:42.189236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exclude a specific column, for example 'category2'\ncolumns_to_plot = features_data.columns.drop('feature')\n\n# Loop through each categorical column and create a pie chart\nfor col in columns_to_plot:\n    plt.figure(figsize=(6, 6))\n    features_data[col].value_counts().plot.pie(autopct='%1.1f%%', figsize=(5, 5))\n    plt.title(f'Distribution of {col}')  # Add title for each pie chart\n    plt.ylabel('')  # Removes the default y-label\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:42.193391Z","iopub.execute_input":"2024-10-17T17:00:42.193952Z","iopub.status.idle":"2024-10-17T17:00:45.520498Z","shell.execute_reply.started":"2024-10-17T17:00:42.193891Z","shell.execute_reply":"2024-10-17T17:00:45.518207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in features_data.columns:\n    features_data[col_name]=labelencoder.fit_transform(features_data[col_name]).astype(int)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:45.523626Z","iopub.execute_input":"2024-10-17T17:00:45.524906Z","iopub.status.idle":"2024-10-17T17:00:45.557316Z","shell.execute_reply.started":"2024-10-17T17:00:45.524806Z","shell.execute_reply":"2024-10-17T17:00:45.554597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = features_data.corr()\n\n# Create the heatmap with seaborn\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\n\n# Display the heatmap\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:45.560090Z","iopub.execute_input":"2024-10-17T17:00:45.562287Z","iopub.status.idle":"2024-10-17T17:00:46.989588Z","shell.execute_reply.started":"2024-10-17T17:00:45.562180Z","shell.execute_reply":"2024-10-17T17:00:46.988098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responders_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:46.991187Z","iopub.execute_input":"2024-10-17T17:00:46.991773Z","iopub.status.idle":"2024-10-17T17:00:47.015635Z","shell.execute_reply.started":"2024-10-17T17:00:46.991703Z","shell.execute_reply":"2024-10-17T17:00:47.014118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"responders_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:47.017855Z","iopub.execute_input":"2024-10-17T17:00:47.018453Z","iopub.status.idle":"2024-10-17T17:00:47.031896Z","shell.execute_reply.started":"2024-10-17T17:00:47.018389Z","shell.execute_reply":"2024-10-17T17:00:47.030092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Exclude a specific column, for example 'category2'\ncolumns_to_plot = responders_data.columns.drop('responder')\n\n# Loop through each categorical column and create a pie chart\nfor col in columns_to_plot:\n    plt.figure(figsize=(6, 6))\n    responders_data[col].value_counts().plot.pie(autopct='%1.1f%%', figsize=(5, 5))\n    plt.title(f'Distribution of {col}')  # Add title for each pie chart\n    plt.ylabel('')  # Removes the default y-label\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:47.034929Z","iopub.execute_input":"2024-10-17T17:00:47.035793Z","iopub.status.idle":"2024-10-17T17:00:47.896798Z","shell.execute_reply.started":"2024-10-17T17:00:47.035721Z","shell.execute_reply":"2024-10-17T17:00:47.895086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#  object datatype columns encoding:\nfrom sklearn.preprocessing import LabelEncoder\nlabelencoder = LabelEncoder()\nfor col_name in responders_data.columns:\n    responders_data[col_name]=labelencoder.fit_transform(responders_data[col_name]).astype(int)\n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:47.899192Z","iopub.execute_input":"2024-10-17T17:00:47.900514Z","iopub.status.idle":"2024-10-17T17:00:47.917490Z","shell.execute_reply.started":"2024-10-17T17:00:47.900425Z","shell.execute_reply":"2024-10-17T17:00:47.915307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = responders_data.corr()\n\n# Create the heatmap with seaborn\nplt.figure(figsize=(10, 8))\nsns.heatmap(corr, annot=True, cmap='coolwarm', fmt=\".2f\")\n\n# Display the heatmap\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:47.928606Z","iopub.execute_input":"2024-10-17T17:00:47.930874Z","iopub.status.idle":"2024-10-17T17:00:48.434691Z","shell.execute_reply.started":"2024-10-17T17:00:47.930763Z","shell.execute_reply":"2024-10-17T17:00:48.433483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:48.436323Z","iopub.execute_input":"2024-10-17T17:00:48.436743Z","iopub.status.idle":"2024-10-17T17:00:48.451485Z","shell.execute_reply.started":"2024-10-17T17:00:48.436699Z","shell.execute_reply":"2024-10-17T17:00:48.450073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:48.453479Z","iopub.execute_input":"2024-10-17T17:00:48.454270Z","iopub.status.idle":"2024-10-17T17:00:48.868302Z","shell.execute_reply.started":"2024-10-17T17:00:48.454216Z","shell.execute_reply":"2024-10-17T17:00:48.867035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/train.parquet\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:00:48.869897Z","iopub.execute_input":"2024-10-17T17:00:48.870393Z","iopub.status.idle":"2024-10-17T17:01:30.201792Z","shell.execute_reply.started":"2024-10-17T17:00:48.870342Z","shell.execute_reply":"2024-10-17T17:01:30.200406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"import pandas as pd\ntrain_data = pd.DataFrame(train_data)\ntrain_data.head()","metadata":{}},{"cell_type":"code","source":"test_data = pl.read_parquet(\"/kaggle/input/jane-street-real-time-market-data-forecasting/test.parquet\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.203577Z","iopub.execute_input":"2024-10-17T17:01:30.204286Z","iopub.status.idle":"2024-10-17T17:01:30.240294Z","shell.execute_reply.started":"2024-10-17T17:01:30.204214Z","shell.execute_reply":"2024-10-17T17:01:30.238922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_data.isnull().sum().sort_values(ascending=False)\ntest_data.isnull().sum().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T01:52:36.634835Z","iopub.execute_input":"2024-10-16T01:52:36.636303Z","iopub.status.idle":"2024-10-16T01:52:36.669358Z","shell.execute_reply.started":"2024-10-16T01:52:36.636248Z","shell.execute_reply":"2024-10-16T01:52:36.667454Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX = train_data.drop(['partition_id','responder_0','responder_1','responder_2','responder_3','responder_4','responder_5','responder_6','responder_7','responder_8'])\ny = train_data['responder_6']\ntest = test_data.drop(['row_id','is_scored'])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.241550Z","iopub.execute_input":"2024-10-17T17:01:30.242052Z","iopub.status.idle":"2024-10-17T17:01:30.253687Z","shell.execute_reply.started":"2024-10-17T17:01:30.241961Z","shell.execute_reply":"2024-10-17T17:01:30.252326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape, test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.255372Z","iopub.execute_input":"2024-10-17T17:01:30.256293Z","iopub.status.idle":"2024-10-17T17:01:30.269352Z","shell.execute_reply.started":"2024-10-17T17:01:30.256239Z","shell.execute_reply":"2024-10-17T17:01:30.267909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split datainto training set and test set\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-16T02:06:27.731575Z","iopub.execute_input":"2024-10-16T02:06:27.731954Z"}}},{"cell_type":"code","source":"print(type(X))\nprint(type(X.sample))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.271094Z","iopub.execute_input":"2024-10-17T17:01:30.271579Z","iopub.status.idle":"2024-10-17T17:01:30.282945Z","shell.execute_reply.started":"2024-10-17T17:01:30.271530Z","shell.execute_reply":"2024-10-17T17:01:30.281611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(y))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.285007Z","iopub.execute_input":"2024-10-17T17:01:30.285604Z","iopub.status.idle":"2024-10-17T17:01:30.299120Z","shell.execute_reply.started":"2024-10-17T17:01:30.285541Z","shell.execute_reply":"2024-10-17T17:01:30.296741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import polars as pl\n\n# Step 1: Sample 10% of the data in Polars\nX_sample = X.sample(fraction=0.1, seed=42)\n\n# Step 2: Get the row indices of the sampled rows\nsample_indices = X_sample.get_column('date_id').to_numpy().astype(int)\n\n# Step 3: Use these indices to sample y\ny_sample = y[sample_indices]\n\n# Display the results\nprint(X_sample)\nprint(y_sample)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:30.301203Z","iopub.execute_input":"2024-10-17T17:01:30.301770Z","iopub.status.idle":"2024-10-17T17:01:43.833189Z","shell.execute_reply.started":"2024-10-17T17:01:30.301706Z","shell.execute_reply":"2024-10-17T17:01:43.831857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert Polars DataFrame to Pandas DataFrame for sklearn compatibility\nX_sample_pd = X_sample.to_pandas()\ny_sample_pd = y_sample.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:43.834776Z","iopub.execute_input":"2024-10-17T17:01:43.835288Z","iopub.status.idle":"2024-10-17T17:01:44.800176Z","shell.execute_reply.started":"2024-10-17T17:01:43.835239Z","shell.execute_reply":"2024-10-17T17:01:44.798851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 4: Define categorical and numerical features\nnumeric_features = X_sample_pd.select_dtypes(include=[np.number]).columns.tolist()\ncategorical_features = X_sample_pd.select_dtypes(exclude=[np.number]).columns.tolist()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:44.802129Z","iopub.execute_input":"2024-10-17T17:01:44.802608Z","iopub.status.idle":"2024-10-17T17:01:47.236771Z","shell.execute_reply.started":"2024-10-17T17:01:44.802556Z","shell.execute_reply":"2024-10-17T17:01:47.235564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 5: Define preprocessors for numeric and categorical features\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),  # Impute missing values\n    ('scaler', StandardScaler())                  # Scale numerical features\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),  # Impute missing values\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))      # One-hot encode categorical features\n])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:47.238551Z","iopub.execute_input":"2024-10-17T17:01:47.239001Z","iopub.status.idle":"2024-10-17T17:01:47.247019Z","shell.execute_reply.started":"2024-10-17T17:01:47.238934Z","shell.execute_reply":"2024-10-17T17:01:47.245570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 6: Combine preprocessors into a ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:47.248693Z","iopub.execute_input":"2024-10-17T17:01:47.249222Z","iopub.status.idle":"2024-10-17T17:01:47.264245Z","shell.execute_reply.started":"2024-10-17T17:01:47.249142Z","shell.execute_reply":"2024-10-17T17:01:47.262829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 7: Define the pipeline with preprocessing and LGBMRegressor\npipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('model', LGBMRegressor(random_state=42))  # LightGBM Regressor\n])","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:47.266027Z","iopub.execute_input":"2024-10-17T17:01:47.266594Z","iopub.status.idle":"2024-10-17T17:01:47.277531Z","shell.execute_reply.started":"2024-10-17T17:01:47.266545Z","shell.execute_reply":"2024-10-17T17:01:47.276161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 8: Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_sample_pd, y_sample_pd, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:47.279481Z","iopub.execute_input":"2024-10-17T17:01:47.279985Z","iopub.status.idle":"2024-10-17T17:01:55.711112Z","shell.execute_reply.started":"2024-10-17T17:01:47.279910Z","shell.execute_reply":"2024-10-17T17:01:55.709857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 9: Fit the pipeline\npipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:01:55.712687Z","iopub.execute_input":"2024-10-17T17:01:55.713100Z","iopub.status.idle":"2024-10-17T17:04:54.832409Z","shell.execute_reply.started":"2024-10-17T17:01:55.713058Z","shell.execute_reply":"2024-10-17T17:04:54.831172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_squared_error, r2_score\n# Get predictions\npredictions = pipeline.predict(X_test)\n# Display metrics\nmse = mean_squared_error(y_test, predictions)\nprint(\"MSE:\", mse)\nrmse = np.sqrt(mse)\nprint(\"RMSE:\", rmse)\nr2 = r2_score(y_test, predictions)\nprint(\"R2:\", r2)\n\n# Plot predicted vs actual\nplt.scatter(y_test, predictions)\nplt.xlabel('Actual Labels')\nplt.ylabel('Predicted Labels')\nplt.title('FloodProbability Predictions')\nz = np.polyfit(y_test, predictions, 1)\np = np.poly1d(z)\nplt.plot(y_test,p(y_test), color='magenta')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:04:54.834284Z","iopub.execute_input":"2024-10-17T17:04:54.834813Z","iopub.status.idle":"2024-10-17T17:05:04.690637Z","shell.execute_reply.started":"2024-10-17T17:04:54.834753Z","shell.execute_reply":"2024-10-17T17:05:04.689195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(test))","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:05:04.692285Z","iopub.execute_input":"2024-10-17T17:05:04.692729Z","iopub.status.idle":"2024-10-17T17:05:04.699769Z","shell.execute_reply.started":"2024-10-17T17:05:04.692682Z","shell.execute_reply":"2024-10-17T17:05:04.698226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you have a Polars DataFrame `test` for making predictions\n# Convert the test DataFrame from Polars to Pandas\nX_test_pd = test.to_pandas()\n\n# Ensure your test DataFrame has the same structure as your training data\n# Use the pipeline to predict\npred = pipeline.predict(X_test_pd)\n\n# Print the predictions\n#print(pred)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:05:04.701520Z","iopub.execute_input":"2024-10-17T17:05:04.702021Z","iopub.status.idle":"2024-10-17T17:05:04.734825Z","shell.execute_reply.started":"2024-10-17T17:05:04.701956Z","shell.execute_reply":"2024-10-17T17:05:04.733430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'row_id': sample_data.row_id, 'responder_6': pred})\n#print(submission.head())\nsubmission.to_parquet('submission.parquet', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-17T17:05:04.736520Z","iopub.execute_input":"2024-10-17T17:05:04.736988Z","iopub.status.idle":"2024-10-17T17:05:04.800228Z","shell.execute_reply.started":"2024-10-17T17:05:04.736923Z","shell.execute_reply":"2024-10-17T17:05:04.799122Z"},"trusted":true},"execution_count":null,"outputs":[]}]}