{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":9667462,"sourceType":"datasetVersion","datasetId":5907276},{"sourceId":9667470,"sourceType":"datasetVersion","datasetId":5907282}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import Necessary Libraries","metadata":{}},{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\"\"\"Forecasting Market Trends Using Machine Learning\n\nAutomatically generated by Colab.\n\nOriginal file is located at\n    https://colab.research.google.com/drive/1idgequAR9F22P2WI0GIqvjbvkTU35yXr\n\n### **Import Necessary Libraries**\n\"\"\"\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_squared_error, r2_score, accuracy_score, classification_report, confusion_matrix\nimport shap\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom sklearn.preprocessing import PolynomialFeatures","metadata":{"_uuid":"e255ea8f-6edb-48e3-b0c7-07d5d3a6eb92","_cell_guid":"6c3e54b7-03e6-40fa-a880-fc61b4cabea6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-10-19T13:26:43.621615Z","iopub.execute_input":"2024-10-19T13:26:43.622084Z","iopub.status.idle":"2024-10-19T13:26:43.630425Z","shell.execute_reply.started":"2024-10-19T13:26:43.622043Z","shell.execute_reply":"2024-10-19T13:26:43.629078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the dataset\ndf=pd.read_csv('/kaggle/input/forecasting-market-trends/features.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.632952Z","iopub.execute_input":"2024-10-19T13:26:43.634011Z","iopub.status.idle":"2024-10-19T13:26:43.643707Z","shell.execute_reply.started":"2024-10-19T13:26:43.633957Z","shell.execute_reply":"2024-10-19T13:26:43.642808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets read the dataset\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.645146Z","iopub.execute_input":"2024-10-19T13:26:43.645922Z","iopub.status.idle":"2024-10-19T13:26:43.677710Z","shell.execute_reply.started":"2024-10-19T13:26:43.645870Z","shell.execute_reply":"2024-10-19T13:26:43.676589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"Data Cleaning","metadata":{}},{"cell_type":"code","source":"# Lets check the shape for dataset\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.680828Z","iopub.execute_input":"2024-10-19T13:26:43.681187Z","iopub.status.idle":"2024-10-19T13:26:43.688277Z","shell.execute_reply.started":"2024-10-19T13:26:43.681144Z","shell.execute_reply":"2024-10-19T13:26:43.687031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking missing/null values\ndf.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.690013Z","iopub.execute_input":"2024-10-19T13:26:43.690520Z","iopub.status.idle":"2024-10-19T13:26:43.704319Z","shell.execute_reply.started":"2024-10-19T13:26:43.690468Z","shell.execute_reply":"2024-10-19T13:26:43.703122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking Duplicates\ndf.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.705957Z","iopub.execute_input":"2024-10-19T13:26:43.706312Z","iopub.status.idle":"2024-10-19T13:26:43.724459Z","shell.execute_reply.started":"2024-10-19T13:26:43.706271Z","shell.execute_reply":"2024-10-19T13:26:43.723109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert all boolean columns to integers (0, 1)\ndf = df.applymap(lambda x: 1 if x is True else (0 if x is False else x))","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.726012Z","iopub.execute_input":"2024-10-19T13:26:43.726390Z","iopub.status.idle":"2024-10-19T13:26:43.735582Z","shell.execute_reply.started":"2024-10-19T13:26:43.726344Z","shell.execute_reply":"2024-10-19T13:26:43.734486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.737181Z","iopub.execute_input":"2024-10-19T13:26:43.737683Z","iopub.status.idle":"2024-10-19T13:26:43.761878Z","shell.execute_reply.started":"2024-10-19T13:26:43.737622Z","shell.execute_reply":"2024-10-19T13:26:43.760601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Feature Selection","metadata":{}},{"cell_type":"code","source":"# Check data types\nprint(df.dtypes)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.763502Z","iopub.execute_input":"2024-10-19T13:26:43.764482Z","iopub.status.idle":"2024-10-19T13:26:43.774395Z","shell.execute_reply.started":"2024-10-19T13:26:43.764428Z","shell.execute_reply":"2024-10-19T13:26:43.773343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Identify non-numeric columns\nnon_numeric_cols = df.select_dtypes(include=['object']).columns.tolist()\nprint(\"Non-numeric columns:\", non_numeric_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.778102Z","iopub.execute_input":"2024-10-19T13:26:43.778454Z","iopub.status.idle":"2024-10-19T13:26:43.788464Z","shell.execute_reply.started":"2024-10-19T13:26:43.778418Z","shell.execute_reply":"2024-10-19T13:26:43.787417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop non-numeric columns if you don't need them for correlation\ndf = df.drop(columns=non_numeric_cols)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.789867Z","iopub.execute_input":"2024-10-19T13:26:43.790186Z","iopub.status.idle":"2024-10-19T13:26:43.803250Z","shell.execute_reply.started":"2024-10-19T13:26:43.790152Z","shell.execute_reply":"2024-10-19T13:26:43.802151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now calculate the correlation\ncorr = df.corr()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.831975Z","iopub.execute_input":"2024-10-19T13:26:43.832616Z","iopub.status.idle":"2024-10-19T13:26:43.840329Z","shell.execute_reply.started":"2024-10-19T13:26:43.832573Z","shell.execute_reply":"2024-10-19T13:26:43.838982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a threshold for correlation\nthreshold = 0.8\nto_drop = []\n# Find features with correlation greater than the threshold\nfor i in range(len(corr.columns)):\n    for j in range(i):\n        if abs(corr.iloc[i, j]) > threshold:\n            colname = corr.columns[i]\n            if colname not in to_drop:\n                to_drop.append(colname)\n# Drop the identified features\ndf.drop(columns=to_drop, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.843160Z","iopub.execute_input":"2024-10-19T13:26:43.843571Z","iopub.status.idle":"2024-10-19T13:26:43.861625Z","shell.execute_reply.started":"2024-10-19T13:26:43.843532Z","shell.execute_reply":"2024-10-19T13:26:43.860428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you want to predict tag_0\ntarget_variable = 'tag_0'\nX = df.drop(columns=[target_variable])\ny = df[target_variable]","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.863034Z","iopub.execute_input":"2024-10-19T13:26:43.863410Z","iopub.status.idle":"2024-10-19T13:26:43.875830Z","shell.execute_reply.started":"2024-10-19T13:26:43.863372Z","shell.execute_reply":"2024-10-19T13:26:43.874116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.877547Z","iopub.execute_input":"2024-10-19T13:26:43.878058Z","iopub.status.idle":"2024-10-19T13:26:43.905819Z","shell.execute_reply.started":"2024-10-19T13:26:43.878007Z","shell.execute_reply":"2024-10-19T13:26:43.904634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.908917Z","iopub.execute_input":"2024-10-19T13:26:43.909297Z","iopub.status.idle":"2024-10-19T13:26:43.972133Z","shell.execute_reply.started":"2024-10-19T13:26:43.909258Z","shell.execute_reply":"2024-10-19T13:26:43.971058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Data Visualization(EDA)","metadata":{}},{"cell_type":"code","source":"# Set the style\nsns.set(style=\"whitegrid\")\n\n# Plot histogram for each feature in the modified DataFrame\ndf.hist(bins=15, figsize=(15, 10), color=('black'))\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:43.973443Z","iopub.execute_input":"2024-10-19T13:26:43.973813Z","iopub.status.idle":"2024-10-19T13:26:49.733427Z","shell.execute_reply.started":"2024-10-19T13:26:43.973764Z","shell.execute_reply":"2024-10-19T13:26:49.731893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Box plot for each feature\nplt.figure(figsize=(15, 10))\nsns.boxplot(data=df, orient=\"h\", palette=\"Set2\")\nplt.title('Boxplot of Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:49.735130Z","iopub.execute_input":"2024-10-19T13:26:49.736020Z","iopub.status.idle":"2024-10-19T13:26:50.405570Z","shell.execute_reply.started":"2024-10-19T13:26:49.735969Z","shell.execute_reply":"2024-10-19T13:26:50.404078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the correlation matrix\ncorr = df.corr()\n\n# Create a heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr, annot=True, fmt=\".2f\", cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:50.407531Z","iopub.execute_input":"2024-10-19T13:26:50.407998Z","iopub.status.idle":"2024-10-19T13:26:51.743901Z","shell.execute_reply.started":"2024-10-19T13:26:50.407941Z","shell.execute_reply":"2024-10-19T13:26:51.742572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Pair plot for the remaining features\nsns.pairplot(df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:26:51.745291Z","iopub.execute_input":"2024-10-19T13:26:51.745664Z","iopub.status.idle":"2024-10-19T13:28:43.647095Z","shell.execute_reply.started":"2024-10-19T13:26:51.745625Z","shell.execute_reply":"2024-10-19T13:28:43.645664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scatter plot between two specific tags, e.g., tag_0 and tag_1\nplt.figure(figsize=(10, 6))\nsns.scatterplot(data=df, x='tag_0', y='tag_1', alpha=0.7)\nplt.title('Scatter Plot between Tag 0 and Tag 1')\nplt.xlabel('Tag 0')\nplt.ylabel('Tag 1')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:43.648636Z","iopub.execute_input":"2024-10-19T13:28:43.649033Z","iopub.status.idle":"2024-10-19T13:28:44.107320Z","shell.execute_reply.started":"2024-10-19T13:28:43.648992Z","shell.execute_reply":"2024-10-19T13:28:44.106230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Violin plot for tag_0\nplt.figure(figsize=(18, 16))\nsns.violinplot(data=df, x=df.index, y='tag_0')\nplt.title('Violin Plot of Tag 0')\nplt.xlabel('Index')\nplt.ylabel('Tag 0')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:44.108679Z","iopub.execute_input":"2024-10-19T13:28:44.109030Z","iopub.status.idle":"2024-10-19T13:28:45.590282Z","shell.execute_reply.started":"2024-10-19T13:28:44.108993Z","shell.execute_reply":"2024-10-19T13:28:45.589072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming you want to plot tag_0 over its index\nplt.figure(figsize=(10, 6))\nplt.plot(df.index, df['tag_0'], label='Tag 0', color='blue')\nplt.title('Line Plot of Tag 0 over Index')\nplt.xlabel('Index')\nplt.ylabel('Tag 0')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:45.591819Z","iopub.execute_input":"2024-10-19T13:28:45.592206Z","iopub.status.idle":"2024-10-19T13:28:46.086724Z","shell.execute_reply.started":"2024-10-19T13:28:45.592168Z","shell.execute_reply":"2024-10-19T13:28:46.085583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# Create bins for tag_0 (you can adjust the bins as necessary)\nbins = [0, 0.5, 1, 2, 3, 4, 5]  # Define your bins\nlabels = ['0-0.5', '0.5-1', '1-2', '2-3', '3-4', '4-5']  # Define labels for the bins\n\n# Cut the data into bins\ntag_0_binned = pd.cut(df['tag_0'], bins=bins, labels=labels)\n\n# Get counts of each bin\ntag_0_counts = tag_0_binned.value_counts()\n\n# Create pie chart\nplt.figure(figsize=(12, 10))\nplt.pie(tag_0_counts, labels=tag_0_counts.index, autopct='%1.1f%%', startangle=140)\nplt.title('Distribution of Tag 0 Binned')\nplt.axis('equal')  # Equal aspect ratio ensures the pie chart is circular.\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.088171Z","iopub.execute_input":"2024-10-19T13:28:46.088542Z","iopub.status.idle":"2024-10-19T13:28:46.334083Z","shell.execute_reply.started":"2024-10-19T13:28:46.088496Z","shell.execute_reply":"2024-10-19T13:28:46.332773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Model Evaluation","metadata":{}},{"cell_type":"code","source":"# Assuming 'tag_0' is our target variable and the rest are features\nX = df.drop(columns=['tag_0'])\ny = df['tag_0']\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.340650Z","iopub.execute_input":"2024-10-19T13:28:46.341525Z","iopub.status.idle":"2024-10-19T13:28:46.353318Z","shell.execute_reply.started":"2024-10-19T13:28:46.341482Z","shell.execute_reply":"2024-10-19T13:28:46.352040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check shapes of training and testing sets\nprint(\"Shape of X_train:\", X_train.shape)\nprint(\"Shape of X_test:\", X_test.shape)\nprint(\"Shape of y_train:\", y_train.shape)\nprint(\"Shape of y_test:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.355266Z","iopub.execute_input":"2024-10-19T13:28:46.356197Z","iopub.status.idle":"2024-10-19T13:28:46.364109Z","shell.execute_reply.started":"2024-10-19T13:28:46.356141Z","shell.execute_reply":"2024-10-19T13:28:46.362947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Instantiate and train the model (for regression example)\nmodel = RandomForestRegressor(n_estimators=100, random_state=42)\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.365819Z","iopub.execute_input":"2024-10-19T13:28:46.366680Z","iopub.status.idle":"2024-10-19T13:28:46.579885Z","shell.execute_reply.started":"2024-10-19T13:28:46.366623Z","shell.execute_reply":"2024-10-19T13:28:46.578642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.581031Z","iopub.execute_input":"2024-10-19T13:28:46.581382Z","iopub.status.idle":"2024-10-19T13:28:46.597028Z","shell.execute_reply.started":"2024-10-19T13:28:46.581345Z","shell.execute_reply":"2024-10-19T13:28:46.595694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate evaluation metrics\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error: {mse:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.598474Z","iopub.execute_input":"2024-10-19T13:28:46.598960Z","iopub.status.idle":"2024-10-19T13:28:46.609294Z","shell.execute_reply.started":"2024-10-19T13:28:46.598920Z","shell.execute_reply":"2024-10-19T13:28:46.608093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\n# For regression\ncv_scores = cross_val_score(model, X, y, cv=5, scoring='neg_mean_squared_error')\nprint(f\"Cross-Validated MSE: {-np.mean(cv_scores):.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:46.610923Z","iopub.execute_input":"2024-10-19T13:28:46.611391Z","iopub.status.idle":"2024-10-19T13:28:47.605217Z","shell.execute_reply.started":"2024-10-19T13:28:46.611352Z","shell.execute_reply":"2024-10-19T13:28:47.604117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameter tuning using GridSearchCV\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'max_depth': [None, 10, 20, 30],\n    'min_samples_split': [2, 5, 10],\n}\ngrid_search = GridSearchCV(RandomForestRegressor(random_state=42), param_grid, cv=5, scoring='neg_mean_squared_error')\ngrid_search.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:28:47.606384Z","iopub.execute_input":"2024-10-19T13:28:47.606701Z","iopub.status.idle":"2024-10-19T13:29:28.481465Z","shell.execute_reply.started":"2024-10-19T13:28:47.606667Z","shell.execute_reply":"2024-10-19T13:29:28.480288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Best model from GridSearchCV\nbest_model = grid_search.best_estimator_\nprint(\"Best parameters:\", grid_search.best_params_)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:28.482910Z","iopub.execute_input":"2024-10-19T13:29:28.483242Z","iopub.status.idle":"2024-10-19T13:29:28.489133Z","shell.execute_reply.started":"2024-10-19T13:29:28.483206Z","shell.execute_reply":"2024-10-19T13:29:28.487989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ny_pred = best_model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:28.490668Z","iopub.execute_input":"2024-10-19T13:29:28.491075Z","iopub.status.idle":"2024-10-19T13:29:28.508784Z","shell.execute_reply.started":"2024-10-19T13:29:28.491037Z","shell.execute_reply":"2024-10-19T13:29:28.507602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate evaluation metrics\nmse = mean_squared_error(y_test, y_pred)\nr2 = r2_score(y_test, y_pred)\n\nprint(f\"Mean Squared Error: {mse:.2f}\")\nprint(f\"R² Score: {r2:.2f}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:28.510170Z","iopub.execute_input":"2024-10-19T13:29:28.510579Z","iopub.status.idle":"2024-10-19T13:29:28.521266Z","shell.execute_reply.started":"2024-10-19T13:29:28.510530Z","shell.execute_reply":"2024-10-19T13:29:28.520220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Engineering: Create polynomial features\npoly = PolynomialFeatures(degree=2, include_bias=False)\nX_poly = poly.fit_transform(X)\n\n# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X_poly, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:28.522597Z","iopub.execute_input":"2024-10-19T13:29:28.523052Z","iopub.status.idle":"2024-10-19T13:29:28.539798Z","shell.execute_reply.started":"2024-10-19T13:29:28.523014Z","shell.execute_reply":"2024-10-19T13:29:28.538799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Interpretation using SHAP\nexplainer = shap.Explainer(best_model, X_train)\nshap_values = explainer(X_test)\n\n# Visualize SHAP values\nshap.summary_plot(shap_values, X_test, feature_names=poly.get_feature_names_out(input_features=X.columns))","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:28.541045Z","iopub.execute_input":"2024-10-19T13:29:28.541480Z","iopub.status.idle":"2024-10-19T13:29:29.744922Z","shell.execute_reply.started":"2024-10-19T13:29:28.541430Z","shell.execute_reply":"2024-10-19T13:29:29.743827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming our original DataFrame is 'df' and it has an 'id' column\n# Make sure to store IDs before you split the data\ndf['id'] = df.index\nX = df.drop(columns=[target_variable, 'id'])\ny = df[target_variable]\n\n# Split the data into training and testing sets and keep track of IDs\nX_train, X_test, y_train, y_test, id_train, id_test = train_test_split(\n    X, y, df['id'], test_size=0.2, random_state=42\n)\n\n# Fit your model and make predictions as before\n# ...\n\n# After making predictions (y_pred) for the test set\ny_pred = best_model.predict(X_test)\n\n# Create the submission DataFrame\nsubmission_df = pd.DataFrame({\n    'id': id_test,  \n    'tag_0': y_pred  \n})","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.746483Z","iopub.execute_input":"2024-10-19T13:29:29.746922Z","iopub.status.idle":"2024-10-19T13:29:29.766065Z","shell.execute_reply.started":"2024-10-19T13:29:29.746875Z","shell.execute_reply":"2024-10-19T13:29:29.764883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save the submission DataFrame to CSV\nsubmission_df.to_csv('parquet.csv', index=False)\n\nprint(\"Submission file created: parquet.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.767415Z","iopub.execute_input":"2024-10-19T13:29:29.767813Z","iopub.status.idle":"2024-10-19T13:29:29.787567Z","shell.execute_reply.started":"2024-10-19T13:29:29.767739Z","shell.execute_reply":"2024-10-19T13:29:29.786413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('parquet.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.791428Z","iopub.execute_input":"2024-10-19T13:29:29.791845Z","iopub.status.idle":"2024-10-19T13:29:29.798754Z","shell.execute_reply.started":"2024-10-19T13:29:29.791806Z","shell.execute_reply":"2024-10-19T13:29:29.797800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the saved submission file\nsubmission_df = pd.read_csv('/kaggle/input/parquet/parquet.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.800078Z","iopub.execute_input":"2024-10-19T13:29:29.800400Z","iopub.status.idle":"2024-10-19T13:29:29.815675Z","shell.execute_reply.started":"2024-10-19T13:29:29.800364Z","shell.execute_reply":"2024-10-19T13:29:29.814580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the contents of the submission file\nprint(submission_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.818640Z","iopub.execute_input":"2024-10-19T13:29:29.819018Z","iopub.status.idle":"2024-10-19T13:29:29.827012Z","shell.execute_reply.started":"2024-10-19T13:29:29.818980Z","shell.execute_reply":"2024-10-19T13:29:29.825835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-19T13:29:29.828720Z","iopub.execute_input":"2024-10-19T13:29:29.829137Z","iopub.status.idle":"2024-10-19T13:29:29.843288Z","shell.execute_reply.started":"2024-10-19T13:29:29.829099Z","shell.execute_reply":"2024-10-19T13:29:29.842200Z"},"trusted":true},"execution_count":null,"outputs":[]}]}