{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":202917725,"sourceType":"kernelVersion"}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Super quick\n## V1: CSV data only.\n- Dropped any rows where the target = nan\n- Really quick EDA of distribution of target\n- Used target mean encodes for LGBM\n- One LGBM model, basic params for training\n- 5 fold CV\n\n# V2: submission\n- Trained LGBM on full dataset and made a submisison\n- LB score = 0.326 V4\n\n# V3: custom objective\nUpdates:\n- Used the custom QWK method as the objective of the model\n- Fixed handling of NaNs in my mean target encoding utility function (now just uses NaN as its own category)\n- Bad score = 0.012 !! \n\n# V4: improve score?\n- Use LGBM regular categorical handling, stop doing target encodes","metadata":{}},{"cell_type":"code","source":"import utility as utl \n# My little collection of useful functions, updated periodically...\n# https://www.kaggle.com/code/beezus666/utility\nimport numpy as np\nimport pandas as pd\nimport os\nimport plotly.express as px\nimport warnings\nwarnings.filterwarnings('ignore')\ntraining = False","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:00:13.069838Z","iopub.execute_input":"2024-11-03T17:00:13.070938Z","iopub.status.idle":"2024-11-03T17:00:13.077079Z","shell.execute_reply.started":"2024-11-03T17:00:13.070887Z","shell.execute_reply":"2024-11-03T17:00:13.075876Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_csv_files_to_pandas(directory, date_field):\n    for dirname, _, filenames in os.walk(directory):\n        for filename in filenames:\n            if filename.endswith('.csv'):\n                print(os.path.join(dirname, filename))\n                no_ext = f'{os.path.splitext(filename)[0]}_df'\n                no_ext = no_ext.replace(\" \", \"_\")\n                no_ext = no_ext.replace(\"-\", \"_\")\n                \n                # Read the CSV headers first to check for the date_field\n                temp_df = pd.read_csv(os.path.join(dirname, filename), nrows=0)\n                if date_field in temp_df.columns:\n                    # Parse dates if the date_field is present, but don't set it as the index\n                    globals()[no_ext] = pd.read_csv(\n                        os.path.join(dirname, filename), \n                        parse_dates=[date_field],\n                        index_col=False  # Ensure the date field is not set as the index\n                    )\n                else:\n                    # Read normally without parsing dates\n                    globals()[no_ext] = pd.read_csv(os.path.join(dirname, filename), index_col=False)\n                del temp_df\n                utl.df_info(globals()[no_ext], no_ext)\n\nload_csv_files_to_pandas('/kaggle/input', 'date_field_name')","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:47:33.751217Z","iopub.execute_input":"2024-11-03T16:47:33.752116Z","iopub.status.idle":"2024-11-03T16:47:36.428172Z","shell.execute_reply.started":"2024-11-03T16:47:33.752053Z","shell.execute_reply":"2024-11-03T16:47:36.426929Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature engineering\n\n* Drop rows where sii = nan\n* Target mean encode any categorical column\n\nAs I mentioned, extra fast here... just want to see what's important features and get a baseline to improve on...","metadata":{}},{"cell_type":"code","source":"train_df_cleaned = (\n    train_df\n    .pipe(lambda df: df.dropna(subset=['sii']))  # Drop rows where 'sii' is NaN initially\n    .pipe(lambda df: df.drop(columns=set(df.columns) - set(test_df.columns) - {'sii'}))  # Ensure 'sii' is retained\n    .pipe(lambda df: df.assign(**{col: df[col].astype('category') for col in utl.get_columns_by_type(df, 'object') if col != 'id'}))\n    .pipe(lambda df: df.dropna(subset=['sii']))  # Drop rows where 'sii' is NaN again, if any reappear\n    .pipe(lambda df: df.drop(columns=['id']))  # Drop the 'id' column\n\n)\n\nutl.summary(train_df_cleaned)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:35.683599Z","iopub.execute_input":"2024-11-03T16:58:35.684105Z","iopub.status.idle":"2024-11-03T16:58:35.896962Z","shell.execute_reply.started":"2024-11-03T16:58:35.684058Z","shell.execute_reply":"2024-11-03T16:58:35.895785Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quick plot\n\nJust looking at the distro of the target variable","metadata":{}},{"cell_type":"code","source":"# Create a count of each value in the 'sii' column\nif training:\n    sii_counts = train_df['sii'].value_counts().sort_index().reset_index()\n    sii_counts.columns = ['sii', 'count']\n    \n    # Create the plot\n    fig = px.bar(sii_counts, x='sii', y='count', title='Distribution of sii Values')\n    fig.update_layout(\n        xaxis_title='sii Values',\n        yaxis_title='Count',\n        xaxis = dict(tickmode='linear'),  # Ensure only integer ticks\n        template='plotly_white'\n    )\n    \n    fig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:39.430686Z","iopub.execute_input":"2024-11-03T16:58:39.431180Z","iopub.status.idle":"2024-11-03T16:58:42.020174Z","shell.execute_reply.started":"2024-11-03T16:58:39.431134Z","shell.execute_reply":"2024-11-03T16:58:42.018908Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quick LGBM train\n\n* Set up CV\n* Set up def for evaluation function.","metadata":{}},{"cell_type":"code","source":"y = train_df_cleaned.pop(\"sii\").astype(int)\na = y.mean()\nb = y.var(ddof=0)\n\ny_min = y.min()\ny_max = y.max()","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:48.862549Z","iopub.execute_input":"2024-11-03T16:58:48.862976Z","iopub.status.idle":"2024-11-03T16:58:48.871299Z","shell.execute_reply.started":"2024-11-03T16:58:48.862935Z","shell.execute_reply":"2024-11-03T16:58:48.869876Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import cohen_kappa_score\nimport numpy as np\n\n# Define the custom evaluation metric for QWK\ndef quadratic_weighted_kappa(preds, data):\n    y_true = data.get_label()\n    y_pred = preds.clip(y_min, y_max).round()  # Clip and round the predictions\n    qwk = cohen_kappa_score(y_true, y_pred, weights=\"quadratic\")\n    return 'QWK', qwk, True\n\n'''def qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)  # Clip predictions between y_min and y_max\n\n    grad = preds - labels  # Simple difference\n    hess = np.ones_like(grad)  # Hessian is constant 1\n    \n    return grad, hess\n'''\n# much thanks...\n# https://www.kaggle.com/code/rsakata/cmi-piu-optimize-qwk-by-lgb/notebook\ndef qwk_obj(preds, dtrain):\n    labels = dtrain.get_label()\n    preds = preds.clip(y_min, y_max)\n    f = 1/2 * np.sum((preds - labels)**2)\n    g = 1/2 * np.sum((preds - a)**2 + b)\n    df = preds - labels\n    dg = preds - a\n    grad = (df/g - f*dg/g**2)*len(labels)\n    hess = np.ones(len(labels))\n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:49.872219Z","iopub.execute_input":"2024-11-03T16:58:49.872666Z","iopub.status.idle":"2024-11-03T16:58:51.396092Z","shell.execute_reply.started":"2024-11-03T16:58:49.872622Z","shell.execute_reply":"2024-11-03T16:58:51.394834Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize the score with a fixed value (optional, may not add back to predictions)\ninit_score = 2.0\n\n# Initialize StratifiedKFold\nn_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# Set the model parameters\n'''params = {\n    'objective': qwk_obj,  # Custom objective for QWK\n    'metric': 'None',  # No default metric; we'll use custom evaluation\n    'learning_rate': 0.1,  # Increased learning rate to allow faster learning\n    'num_leaves': 16,\n    'feature_fraction': 0.5,\n    'verbose': -1\n}\n'''\n\nparams = {\n    \"objective\": qwk_obj,\n    \"metric\": \"None\",\n    \"verbosity\": -1,\n    \"learning_rate\": 0.01,\n    \"num_leaves\": 24,\n    \"feature_fraction\": 0.5\n}","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:51.398129Z","iopub.execute_input":"2024-11-03T16:58:51.398793Z","iopub.status.idle":"2024-11-03T16:58:51.405865Z","shell.execute_reply.started":"2024-11-03T16:58:51.398749Z","shell.execute_reply":"2024-11-03T16:58:51.404591Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\n# Define StratifiedKFold\nn_splits = 5\nkf = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\n# Prepare the LightGBM dataset with init_score\ntrain_data = lgb.Dataset(train_df_cleaned, label=y, init_score=[init_score] * len(train_df_cleaned))\n\n# Get the custom train/validation indices for each fold\nfolds = [(train_idx, val_idx) for train_idx, val_idx in kf.split(train_df_cleaned, y)]\n\n# Use lgb.cv for cross-validation with custom folds\n\nif training:\n    cv_results = lgb.cv(\n        params=params,\n        train_set=train_data,  # The full dataset\n        num_boost_round=10000,  # Set a high number of rounds; early stopping will limit it\n        folds=folds,  # Pass the custom folds from StratifiedKFold\n        feval=quadratic_weighted_kappa,  # Custom QWK evaluation metric\n        callbacks=[\n            lgb.early_stopping(stopping_rounds=100, verbose=True),\n            lgb.log_evaluation(100)\n        ],\n        return_cvbooster=True\n    )\n    \n    # Capture the best iteration based on early stopping\n    best_iteration = len(cv_results['valid QWK-mean'])\n    print(f\"Best iteration from CV: {best_iteration}\")\n    \n    # Extract models for each fold if needed\n    models = cv_results[\"cvbooster\"].boosters","metadata":{"execution":{"iopub.status.busy":"2024-11-03T16:58:52.977291Z","iopub.execute_input":"2024-11-03T16:58:52.977760Z","iopub.status.idle":"2024-11-03T16:59:28.350791Z","shell.execute_reply.started":"2024-11-03T16:58:52.977713Z","shell.execute_reply":"2024-11-03T16:59:28.349379Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if training:\n    import plotly.express as px\n    import pandas as pd\n    import numpy as np\n    \n    # Initialize a DataFrame to store feature importances\n    all_importances = []\n    \n    # Loop through each model to get feature importances\n    for model in models:\n        feature_importances = model.feature_importance(importance_type='gain')  # You can also use 'split' for counts\n        feature_names = model.feature_name()\n        \n        # Store the importance as a DataFrame for each model\n        all_importances.append(pd.DataFrame({'feature': feature_names, 'importance': feature_importances}))\n    \n    # Concatenate all importances into a single DataFrame\n    importance_df = pd.concat(all_importances)\n    \n    # Group by feature name and compute the average importance across models\n    mean_importance_df = importance_df.groupby('feature', as_index=False).agg({'importance': 'mean'})\n    \n    # Sort the DataFrame by average importance and select the top 10 features\n    top_10_features = mean_importance_df.sort_values(by='importance', ascending=False).head(10)\n    \n    # Create a bar plot using plotly express\n    fig = px.bar(top_10_features, x='importance', y='feature', orientation='h', \n                 title='Top 10 Most Important Features (Average Across Models)', \n                 labels={'importance': 'Average Feature Importance', 'feature': 'Feature Name'})\n    \n    # Show the plot\n    fig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:00:38.045800Z","iopub.execute_input":"2024-11-03T17:00:38.046299Z","iopub.status.idle":"2024-11-03T17:00:38.147842Z","shell.execute_reply.started":"2024-11-03T17:00:38.046252Z","shell.execute_reply":"2024-11-03T17:00:38.145921Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if training:\n    top_20 = mean_importance_df.sort_values(by='importance', ascending=False).head(20)\n    top_20.feature.values\n    top_20['importance'] = top_20['importance'].round(0).astype(int)\n    top_20","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:00:44.986756Z","iopub.execute_input":"2024-11-03T17:00:44.987202Z","iopub.status.idle":"2024-11-03T17:00:45.009136Z","shell.execute_reply.started":"2024-11-03T17:00:44.987161Z","shell.execute_reply":"2024-11-03T17:00:45.007615Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# use all data and submit\n","metadata":{}},{"cell_type":"code","source":"test_df = (\n    test_df\n    .pipe(lambda df: df.assign(**{col: df[col].astype('category') for col in utl.get_columns_by_type(df, 'object') if col != 'id'}))    \n    #.pipe(lambda df: df.drop(columns=['id']))  # Drop the 'id' column\n\n)\n\nutl.summary(test_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-03T17:05:21.014333Z","iopub.execute_input":"2024-11-03T17:05:21.014749Z","iopub.status.idle":"2024-11-03T17:05:21.177989Z","shell.execute_reply.started":"2024-11-03T17:05:21.014706Z","shell.execute_reply":"2024-11-03T17:05:21.176784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train the final model on the entire dataset using the best iteration from CV\nif training == False:\n    best_iteration = 519\nfinal_model = lgb.train(\n    params=params,\n    train_set=train_data,\n    num_boost_round=best_iteration,  # Use the optimal number of boosting rounds from CV\n    feval=quadratic_weighted_kappa,  # Custom evaluation metric if needed\n    callbacks=[lgb.log_evaluation(period=100)]  # Log evaluation every 100 rounds\n)\n\nprint(\"Final model training complete.\")","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:14:00.119103Z","iopub.execute_input":"2024-11-03T17:14:00.119691Z","iopub.status.idle":"2024-11-03T17:14:05.074162Z","shell.execute_reply.started":"2024-11-03T17:14:00.119640Z","shell.execute_reply":"2024-11-03T17:14:05.072841Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make predictions on the test set\ny_test_pred = final_model.predict(test_df.drop(columns=['id']))\n\n# Convert predictions to class labels (0, 1, 2, 3) by rounding to the nearest integer\ny_test_pred_labels = np.round(y_test_pred).astype(int)\n\n# Clip to ensure the values are within the expected class range (0 to 3)\ny_test_pred_labels = np.clip(y_test_pred_labels, 0, 3)\n\n# Prepare the submission file\nsubmission = test_df[['id']].copy()  # Copy 'id' column for submission\nsubmission['prediction'] = y_test_pred_labels\n\n# Save predictions to a CSV file\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:05:39.593253Z","iopub.execute_input":"2024-11-03T17:05:39.593691Z","iopub.status.idle":"2024-11-03T17:05:39.624056Z","shell.execute_reply.started":"2024-11-03T17:05:39.593647Z","shell.execute_reply":"2024-11-03T17:05:39.622273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"utl.df_info(submission, 'sub')","metadata":{"execution":{"iopub.status.busy":"2024-11-03T17:05:44.062375Z","iopub.execute_input":"2024-11-03T17:05:44.062888Z","iopub.status.idle":"2024-11-03T17:05:44.075230Z","shell.execute_reply.started":"2024-11-03T17:05:44.062841Z","shell.execute_reply":"2024-11-03T17:05:44.073754Z"},"trusted":true},"outputs":[],"execution_count":null}]}