{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9871156,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Table of Contents\n* [Load Data](#load)\n* [Weights](#weights)\n* [Features](#features)\n* [Targets](#targets)\n* [Distribution fits on target](#dist_fit)\n    * [Feature Scoring (Mutual Info & Spearmen & Tail Similarity)](#feature_score)\n    * [Comparision Visualization](#compare_viz)\n    * [Score by Group](#score_group)\n* [Correlation Features vs Features](#corr_features_features)\n* [Correlation Targets vs Features](#corr_target_features)","metadata":{}},{"cell_type":"code","source":"# install package for distribution fitting\n!pip install fitter","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-11-04T05:36:27.166583Z","iopub.execute_input":"2024-11-04T05:36:27.166977Z","iopub.status.idle":"2024-11-04T05:36:44.413553Z","shell.execute_reply.started":"2024-11-04T05:36:27.166937Z","shell.execute_reply":"2024-11-04T05:36:44.412258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# packages\n\n# standard\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport time\n\n# garbage collection\nimport gc\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# stats\nimport random\nimport scipy.stats\nfrom fitter import Fitter, get_common_distributions, get_distributions\n\n# loading\nfrom tqdm import tqdm\n\n# other stuff\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-04T07:26:15.988406Z","iopub.execute_input":"2024-11-04T07:26:15.988896Z","iopub.status.idle":"2024-11-04T07:26:15.996533Z","shell.execute_reply.started":"2024-11-04T07:26:15.988848Z","shell.execute_reply":"2024-11-04T07:26:15.995232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show files\n!ls -l '../input/jane-street-real-time-market-data-forecasting'","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:36:56.481882Z","iopub.execute_input":"2024-11-04T05:36:56.482489Z","iopub.status.idle":"2024-11-04T05:36:57.671441Z","shell.execute_reply.started":"2024-11-04T05:36:56.482438Z","shell.execute_reply":"2024-11-04T05:36:57.669877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# details - training data\n!ls -lR '../input/jane-street-real-time-market-data-forecasting/train.parquet'","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:36:59.963532Z","iopub.execute_input":"2024-11-04T05:36:59.963971Z","iopub.status.idle":"2024-11-04T05:37:01.158168Z","shell.execute_reply.started":"2024-11-04T05:36:59.963928Z","shell.execute_reply":"2024-11-04T05:37:01.156860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\npd.set_option('display.max_rows', 100) # increase number of displayed rows\n\n# random seed\nmy_random_seed = 123\nrandom.seed(128)\n\n# aesthetics\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:04.350614Z","iopub.execute_input":"2024-11-04T05:37:04.351092Z","iopub.status.idle":"2024-11-04T05:37:04.358494Z","shell.execute_reply.started":"2024-11-04T05:37:04.351034Z","shell.execute_reply":"2024-11-04T05:37:04.357200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='load'></a>\n# Load Data","metadata":{}},{"cell_type":"code","source":"# load data\nt1 = time.time()\ndf_0 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=0/part-0.parquet')\ndf_1 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=1/part-0.parquet')\ndf_2 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=2/part-0.parquet')\ndf_3 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=3/part-0.parquet')\ndf_4 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=4/part-0.parquet')\ndf_5 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=5/part-0.parquet')\ndf_6 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=6/part-0.parquet')\ndf_7 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=7/part-0.parquet')\ndf_8 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=8/part-0.parquet')\ndf_9 = pl.read_parquet('../input/jane-street-real-time-market-data-forecasting/train.parquet/partition_id=9/part-0.parquet')\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1,4))","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:07.047581Z","iopub.execute_input":"2024-11-04T05:37:07.048032Z","iopub.status.idle":"2024-11-04T05:37:52.549987Z","shell.execute_reply.started":"2024-11-04T05:37:07.047988Z","shell.execute_reply":"2024-11-04T05:37:52.548714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine in one data frame\ndf = pl.concat([df_0,df_1,df_2,df_3,df_4,df_5,df_6,df_7,df_8,df_9])","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:52.552982Z","iopub.execute_input":"2024-11-04T05:37:52.553451Z","iopub.status.idle":"2024-11-04T05:37:52.581737Z","shell.execute_reply.started":"2024-11-04T05:37:52.553408Z","shell.execute_reply":"2024-11-04T05:37:52.580268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean up\ndel df_0,df_1,df_2,df_3,df_4,df_5,df_6,df_7,df_8,df_9\ngc.collect();","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:52.583801Z","iopub.execute_input":"2024-11-04T05:37:52.584269Z","iopub.status.idle":"2024-11-04T05:37:52.706668Z","shell.execute_reply.started":"2024-11-04T05:37:52.584222Z","shell.execute_reply":"2024-11-04T05:37:52.705355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preview\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:52.709120Z","iopub.execute_input":"2024-11-04T05:37:52.709767Z","iopub.status.idle":"2024-11-04T05:37:52.756982Z","shell.execute_reply.started":"2024-11-04T05:37:52.709723Z","shell.execute_reply":"2024-11-04T05:37:52.755940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data frame size\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:37:52.758375Z","iopub.execute_input":"2024-11-04T05:37:52.758742Z","iopub.status.idle":"2024-11-04T05:37:52.765522Z","shell.execute_reply.started":"2024-11-04T05:37:52.758703Z","shell.execute_reply":"2024-11-04T05:37:52.764358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='features'></a>\n# Features","metadata":{}},{"cell_type":"code","source":"features = ['feature_' + str(x).zfill(2) for x in range(78+1)]\nprint(features)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:04.684529Z","iopub.execute_input":"2024-11-04T05:38:04.684964Z","iopub.status.idle":"2024-11-04T05:38:04.691096Z","shell.execute_reply.started":"2024-11-04T05:38:04.684922Z","shell.execute_reply":"2024-11-04T05:38:04.689855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='targets'></a>\n# Targets","metadata":{}},{"cell_type":"code","source":"# target names\ntargets = ['responder_' + str(x) for x in range(8+1)]\nprint(targets)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:12.570655Z","iopub.execute_input":"2024-11-04T05:38:12.571158Z","iopub.status.idle":"2024-11-04T05:38:12.578366Z","shell.execute_reply.started":"2024-11-04T05:38:12.571111Z","shell.execute_reply":"2024-11-04T05:38:12.576909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 💡 Looks like an artificial cap at -/+5","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8,3))\nplt.hist(df['responder_6'], bins=100, color=default_color_3)\nplt.grid(color='black', linestyle='-', linewidth=0.5)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:22.284980Z","iopub.execute_input":"2024-11-04T05:38:22.286293Z","iopub.status.idle":"2024-11-04T05:38:23.716295Z","shell.execute_reply.started":"2024-11-04T05:38:22.286244Z","shell.execute_reply":"2024-11-04T05:38:23.715133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='dist_fit'></a>\n# Distribution fits on target","metadata":{}},{"cell_type":"code","source":"# this is our actual target\ntarget = 'responder_6'","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:27.180447Z","iopub.execute_input":"2024-11-04T05:38:27.180890Z","iopub.status.idle":"2024-11-04T05:38:27.186008Z","shell.execute_reply.started":"2024-11-04T05:38:27.180846Z","shell.execute_reply":"2024-11-04T05:38:27.184650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='feature_score'></a>\n## Feature Scoring (Mutual Info & Spearmen & Tail Similarity)","metadata":{}},{"cell_type":"code","source":"sample_x_y = df.sample(500000, seed=my_random_seed)\ny = sample_x_y['responder_6']\n\n# Get X\nfeature_cols = [col for col in sample_x_y.columns if col.startswith('feature_')]\nX = sample_x_y[feature_cols]\n\n# Fill na with mean\nX = X.fill_null(strategy='mean')","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:43.459455Z","iopub.execute_input":"2024-11-04T05:38:43.459893Z","iopub.status.idle":"2024-11-04T05:38:49.379708Z","shell.execute_reply.started":"2024-11-04T05:38:43.459852Z","shell.execute_reply":"2024-11-04T05:38:49.378767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.select(pl.all().has_nulls())","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:38:50.952356Z","iopub.execute_input":"2024-11-04T05:38:50.952916Z","iopub.status.idle":"2024-11-04T05:38:50.983118Z","shell.execute_reply.started":"2024-11-04T05:38:50.952852Z","shell.execute_reply":"2024-11-04T05:38:50.981784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport polars as pl\nfrom scipy import stats\nfrom sklearn.feature_selection import mutual_info_regression\n\nclass FeatureEvaluator:\n    def __init__(self, target_dist_type='johnsonsu', winsorize=True, winsorize_quantiles=(0.05, 0.95)):\n        self.target_dist_type = target_dist_type\n        self.winsorize = winsorize\n        self.winsorize_quantiles = winsorize_quantiles\n        self.original_params = {}\n        self.winsorized_params = {}\n    \n    def evaluate_features(self, X, y):\n        \"\"\"\n        Evaluate features for both original and winsorized data\n        \"\"\"\n        evaluations = {\n            'original': {},\n            'winsorized': {} if self.winsorize else None\n        }\n        \n        y_np = y.to_numpy()\n        \n        # Process original data\n        print(\"\\nProcessing original data:\")\n        for column in tqdm(X.columns, desc=\"Evaluating original features\"):\n            feature_np = X[column].to_numpy()\n            # Store original evaluation results\n            evaluations['original'][column] = self._evaluate_single_feature(\n                feature_np, y_np, is_winsorized=False)\n        \n        # Process winsorized data\n        if self.winsorize:\n            print(\"\\nProcessing winsorized data:\")\n            for column in tqdm(X.columns, desc=\"Evaluating winsorized features\"):\n                feature_np = X[column].to_numpy()\n                # Apply winsorization to both feature and target\n                feature_wins = self._winsorize_data(feature_np)\n                y_wins = self._winsorize_data(y_np)\n                # Store winsorized evaluation results\n                evaluations['winsorized'][column] = self._evaluate_single_feature(\n                    feature_wins, y_wins, is_winsorized=True)\n        \n        return self._generate_final_report(evaluations)\n    \n    def _winsorize_data(self, data):\n        \"\"\"\n        Winsorize the data based on specified quantiles\n        \"\"\"\n        lower, upper = np.percentile(data, \n                                   [self.winsorize_quantiles[0]*100, \n                                    self.winsorize_quantiles[1]*100])\n        return np.clip(data, lower, upper)\n    \n    def _evaluate_single_feature(self, feature, y, is_winsorized):\n        \"\"\"\n        Evaluate a single feature's relationship with the target\n        \"\"\"\n        relationship_stats = self._evaluate_relationship(feature, y)\n        tail_stats = self._evaluate_tail_behavior(feature, y)\n        \n        # Create evaluation dictionary with all metrics\n        evaluation = {\n            'relationship_stats': relationship_stats,\n            'tail_stats': tail_stats,\n            'is_winsorized': is_winsorized\n        }\n        \n        return evaluation\n    \n    def _evaluate_relationship(self, feature, y):\n        \"\"\"\n        Calculate relationship metrics\n        \"\"\"\n        return {\n            'spearman': stats.spearmanr(feature, y)[0],\n            'mutual_info': mutual_info_regression(\n                feature.reshape(-1, 1), y, random_state=42)[0]\n        }\n    \n    def _evaluate_tail_behavior(self, feature, y, percentiles=[0.95]):\n        \"\"\"\n        Calculate tail behavior metrics\n        \"\"\"\n        tail_stats = {}\n        \n        for p in percentiles:\n            feature_threshold = np.percentile(feature, p*100)\n            feature_tail_mask = feature >= feature_threshold\n            \n            y_threshold = np.percentile(y, p*100)\n            y_tail_mask = y >= y_threshold\n            \n            joint_tail_prob = np.mean(feature_tail_mask & y_tail_mask)\n            feature_tail_prob = np.mean(feature_tail_mask)\n            y_tail_prob = np.mean(y_tail_mask)\n            \n            tail_dependence = (joint_tail_prob / (feature_tail_prob * y_tail_prob) \n                             if feature_tail_prob * y_tail_prob > 0 else 0)\n            \n            tail_stats[f'p{int(p*100)}'] = {\n                'tail_dependence': tail_dependence,\n                'joint_tail_prob': joint_tail_prob,\n                'conditional_prob': (joint_tail_prob / feature_tail_prob \n                                    if feature_tail_prob > 0 else 0)\n            }\n        return tail_stats\n    \n    def _calculate_feature_score(self, eval_data):\n        \"\"\"\n        Calculate feature score and store parameters\n        \"\"\"\n        params = {\n            'relationship': {\n                'spearman': abs(eval_data['relationship_stats']['spearman']),\n                'mutual_info': eval_data['relationship_stats']['mutual_info']\n            },\n            'tail': {\n                'tail_dependence': eval_data['tail_stats']['p95']['tail_dependence'],\n                'joint_tail_prob': eval_data['tail_stats']['p95']['joint_tail_prob'],\n                'conditional_prob': eval_data['tail_stats']['p95']['conditional_prob']\n            }\n        }\n        \n        relationship_score = 0.7 * (\n            0.5 * params['relationship']['spearman'] +\n            0.5 * params['relationship']['mutual_info']\n        )\n        \n        tail_score = 0.3 * (\n            0.4 * params['tail']['tail_dependence'] +\n            0.3 * params['tail']['joint_tail_prob'] +\n            0.3 * params['tail']['conditional_prob']\n        )\n        \n        return (relationship_score + tail_score, params)\n    \n    def _rank_features(self, feature_evaluations):\n        \"\"\"\n        Rank features and store parameters\n        \"\"\"\n        scored_features = {}\n        feature_params = {}\n        \n        # Calculate scores and collect parameters\n        for feature, eval_data in feature_evaluations.items():\n            score, params = self._calculate_feature_score(eval_data)\n            scored_features[feature] = score\n            feature_params[feature] = params\n        \n        # Store parameters in appropriate dictionary based on evaluation type\n        if next(iter(feature_evaluations.values()))['is_winsorized']:\n            self.winsorized_params = feature_params\n        else:\n            self.original_params = feature_params\n        \n        ranked_features = sorted(\n            scored_features.items(),\n            key=lambda x: x[1],\n            reverse=True\n        )\n        \n        return {\n            'rankings': ranked_features,\n            'top_features': [f[0] for f in ranked_features[:10]],\n            'feature_groups': self._group_features(feature_evaluations, scored_features)\n        }\n    \n    def _group_features(self, evaluations, scores):\n        \"\"\"\n        Categorizing the range of scores\n        \"\"\"\n        groups = {\n            'strong_predictors': [],\n            'tail_predictors': [],\n            'weak_predictors': []\n        }\n        \n        for feature, eval_data in evaluations.items():\n            score, _ = self._calculate_feature_score(eval_data)\n            tail_stats = eval_data['tail_stats']['p95']\n            \n            if score > 0.6:\n                groups['strong_predictors'].append(feature)\n            elif tail_stats['tail_dependence'] > 1.5:\n                groups['tail_predictors'].append(feature)\n            else:\n                groups['weak_predictors'].append(feature)\n        \n        return groups\n    \n    def _generate_final_report(self, evaluations):\n        report = {\n            'original': self._rank_features(evaluations['original']),\n            'winsorized': self._rank_features(evaluations['winsorized']) if self.winsorize else None,\n            'comparison': self._compare_evaluations(evaluations) if self.winsorize else None\n        }\n        \n        return report\n    \n    def _compare_evaluations(self, evaluations):\n        \"\"\"\n        Comparing Result for between original and winsorized data\n        \"\"\"\n        comparison = {}\n        \n        for feature in evaluations['original'].keys():\n            orig_score, _ = self._calculate_feature_score(evaluations['original'][feature])\n            wins_score, _ = self._calculate_feature_score(evaluations['winsorized'][feature])\n            \n            comparison[feature] = {\n                'original_score': orig_score,\n                'winsorized_score': wins_score,\n                'score_change': wins_score - orig_score,\n                'score_change_pct': (wins_score - orig_score) / orig_score * 100\n            }\n        \n        return comparison\n\ndef print_evaluation_results(evaluation_results):\n    print(\"\\nOriginal Data Results:\")\n    print(\"=\" * 50)\n    _print_single_evaluation(evaluation_results['original'])\n    \n    # pring winsorized result\n    if evaluation_results['winsorized']:\n        print(\"\\nWinsorized Data Results:\")\n        print(\"=\" * 50)\n        _print_single_evaluation(evaluation_results['winsorized'])\n        \n        print(\"\\nComparison Analysis:\")\n        print(\"=\" * 50)\n        _print_comparison(evaluation_results['comparison'])\n\ndef _print_single_evaluation(eval_result):\n    print(\"\\nTop 10 Features:\")\n    print(\"-\" * 40)\n    for feature in eval_result['top_features']:\n        score = next(e[1] for e in eval_result['rankings'] if e[0] == feature)\n        print(f\"{feature}: {score:.4f}\")\n\n    print(\"\\nFeature Groups:\")\n    print(\"-\" * 40)\n    for group, features in eval_result['feature_groups'].items():\n        print(f\"\\n{group.replace('_', ' ').title()}:\")\n        print(f\"Count: {len(features)}\")\n        if features:\n            print(f\"Examples: {', '.join(features[:3])}\")\n\ndef _print_comparison(comparison):\n    sorted_features = sorted(\n        comparison.items(),\n        key=lambda x: abs(x[1]['score_change']),\n        reverse=True\n    )\n    \n    print(\"\\nTop 5 Features with Largest Score Changes:\")\n    print(\"-\" * 40)\n    for feature, stats in sorted_features[:5]:\n        print(f\"{feature}:\")\n        print(f\"  Original Score: {stats['original_score']:.4f}\")\n        print(f\"  Winsorized Score: {stats['winsorized_score']:.4f}\")\n        print(f\"  Change: {stats['score_change']:.4f} ({stats['score_change_pct']:.1f}%)\")\n        \n\ndef print_detailed_parameters(evaluator):\n    \"\"\"\n    Print detailed parameters for both original and winsorized data\n    \"\"\"\n    print(\"\\nOriginal Data Parameters:\")\n    print(\"=\" * 50)\n    _print_params_for_dataset(evaluator.original_params)\n    \n    if evaluator.winsorize:\n        print(\"\\nWinsorized Data Parameters:\")\n        print(\"=\" * 50)\n        _print_params_for_dataset(evaluator.winsorized_params)\n\ndef _print_params_for_dataset(params_dict):\n    for feature, params in params_dict.items():\n        print(f\"\\nFeature: {feature}\")\n        print(\"-\" * 40)\n        print(\"Relationship Parameters:\")\n        print(f\"  Spearman: {params['relationship']['spearman']:.4f}\")\n        print(f\"  Mutual Information: {params['relationship']['mutual_info']:.4f}\")\n        print(\"\\nTail Parameters:\")\n        print(f\"  Tail Dependence: {params['tail']['tail_dependence']:.4f}\")\n        print(f\"  Joint Tail Probability: {params['tail']['joint_tail_prob']:.4f}\")\n        print(f\"  Conditional Probability: {params['tail']['conditional_prob']:.4f}\")\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T08:10:37.648251Z","iopub.execute_input":"2024-11-04T08:10:37.649122Z","iopub.status.idle":"2024-11-04T08:10:37.706368Z","shell.execute_reply.started":"2024-11-04T08:10:37.649042Z","shell.execute_reply":"2024-11-04T08:10:37.704927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluator = FeatureEvaluator(winsorize=True)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T08:10:39.902736Z","iopub.execute_input":"2024-11-04T08:10:39.903837Z","iopub.status.idle":"2024-11-04T08:10:39.908826Z","shell.execute_reply.started":"2024-11-04T08:10:39.903787Z","shell.execute_reply":"2024-11-04T08:10:39.907566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluator.evaluate_features(X, y)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T08:10:42.699511Z","iopub.execute_input":"2024-11-04T08:10:42.699926Z","iopub.status.idle":"2024-11-04T08:31:12.584629Z","shell.execute_reply.started":"2024-11-04T08:10:42.699888Z","shell.execute_reply":"2024-11-04T08:31:12.583545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# evaluation_results = evaluator.evaluate_features(X, y)\n# evaluation_results","metadata":{"execution":{"iopub.status.busy":"2024-11-04T05:59:50.157186Z","iopub.status.idle":"2024-11-04T05:59:50.157682Z","shell.execute_reply.started":"2024-11-04T05:59:50.157447Z","shell.execute_reply":"2024-11-04T05:59:50.157473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(evaluator.original_params)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T08:33:29.447497Z","iopub.execute_input":"2024-11-04T08:33:29.447998Z","iopub.status.idle":"2024-11-04T08:33:29.455402Z","shell.execute_reply.started":"2024-11-04T08:33:29.447949Z","shell.execute_reply":"2024-11-04T08:33:29.453851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(evaluator.winsorized_params)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T07:49:40.356803Z","iopub.execute_input":"2024-11-04T07:49:40.357316Z","iopub.status.idle":"2024-11-04T07:49:40.364916Z","shell.execute_reply.started":"2024-11-04T07:49:40.357261Z","shell.execute_reply":"2024-11-04T07:49:40.363470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Or use the printing function for formatted output\nprint_detailed_parameters(evaluator)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T07:18:08.874313Z","iopub.execute_input":"2024-11-04T07:18:08.875272Z","iopub.status.idle":"2024-11-04T07:18:08.889446Z","shell.execute_reply.started":"2024-11-04T07:18:08.875224Z","shell.execute_reply":"2024-11-04T07:18:08.888164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluation_results","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:17.166966Z","iopub.execute_input":"2024-10-29T05:47:17.167513Z","iopub.status.idle":"2024-10-29T05:47:17.196545Z","shell.execute_reply.started":"2024-10-29T05:47:17.167456Z","shell.execute_reply":"2024-10-29T05:47:17.195211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='compare_viz'></a>\n## Comparision Visualization","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\n\ndef create_comparison_visualizations(evaluation_results):\n    # Set style\n    plt.style.use('seaborn')\n    \n    # Create figure with multiple subplots\n    fig = plt.figure(figsize=(15, 10))\n    gs = fig.add_gridspec(2, 2, height_ratios=[1, 1.2])\n    \n    # 1. Scatter plot of original vs winsorized scores\n    ax1 = fig.add_subplot(gs[0, :])\n    scatter_data = pd.DataFrame([\n        {\n            'feature': feat[0],\n            'original': feat[1],\n            'winsorized': next(w[1] for w in evaluation_results['winsorized']['rankings'] if w[0] == feat[0])\n        }\n        for feat in evaluation_results['original']['rankings']\n    ])\n    \n    ax1.scatter(scatter_data['original'], scatter_data['winsorized'], \n                alpha=0.6, c='blue', label='Features')\n    \n    # Add diagonal line\n    lims = [\n        min(ax1.get_xlim()[0], ax1.get_ylim()[0]),\n        max(ax1.get_xlim()[1], ax1.get_ylim()[1])\n    ]\n    ax1.plot(lims, lims, 'r--', alpha=0.5, label='y=x')\n    \n    ax1.set_xlabel('Original Score')\n    ax1.set_ylabel('Winsorized Score')\n    ax1.set_title('Original vs Winsorized Feature Scores')\n    ax1.legend()\n    \n    # Add correlation coefficient\n    corr = np.corrcoef(scatter_data['original'], scatter_data['winsorized'])[0,1]\n    ax1.text(0.05, 0.95, f'Correlation: {corr:.4f}', \n             transform=ax1.transAxes, bbox=dict(facecolor='white', alpha=0.8))\n    \n    # 2. Top changes bar plot\n    ax2 = fig.add_subplot(gs[1, 0])\n    \n    # Calculate changes\n    changes = pd.DataFrame([\n        {\n            'feature': k,\n            'change': v['score_change_pct']\n        }\n        for k, v in evaluation_results['comparison'].items()\n    ])\n    \n    # Sort by absolute change and get top 10\n    top_changes = changes.nlargest(10, 'change')\n    \n    # Create bar plot\n    bars = ax2.barh(np.arange(len(top_changes)), top_changes['change'])\n    ax2.set_yticks(np.arange(len(top_changes)))\n    ax2.set_yticklabels(top_changes['feature'])\n    ax2.set_xlabel('Score Change (%)')\n    ax2.set_title('Top 10 Score Changes')\n    \n    # Color bars based on positive/negative changes\n    for i, bar in enumerate(bars):\n        if top_changes['change'].iloc[i] > 0:\n            bar.set_color('green')\n        else:\n            bar.set_color('red')\n    \n    # 3. Distribution of changes\n    ax3 = fig.add_subplot(gs[1, 1])\n    sns.histplot(changes['change'], bins=30, ax=ax3)\n    ax3.set_xlabel('Score Change (%)')\n    ax3.set_ylabel('Count')\n    ax3.set_title('Distribution of Score Changes')\n    \n    # Add vertical line at x=0\n    ax3.axvline(x=0, color='r', linestyle='--', alpha=0.5)\n    \n    # Add mean and median annotations\n    mean_change = changes['change'].mean()\n    median_change = changes['change'].median()\n    ax3.text(0.05, 0.95, f'Mean: {mean_change:.4f}%\\nMedian: {median_change:.4f}%', \n             transform=ax3.transAxes, bbox=dict(facecolor='white', alpha=0.8))\n    \n    # Adjust layout\n    plt.tight_layout()\n    return fig\n\n# Create summary stats\ndef print_summary_stats(evaluation_results):\n    changes = pd.DataFrame([\n        {\n            'feature': k,\n            'change': v['score_change_pct'],\n            'original': v['original_score'],\n            'winsorized': v['winsorized_score']\n        }\n        for k, v in evaluation_results['comparison'].items()\n    ])\n    \n    print(\"\\nSummary Statistics:\")\n    print(\"-\" * 50)\n    print(f\"Average absolute change: {changes['change'].abs().mean():.4f}%\")\n    print(f\"Maximum increase: {changes['change'].max():.4f}%\")\n    print(f\"Maximum decrease: {changes['change'].min():.4f}%\")\n    print(f\"Features with >1% change: {(changes['change'].abs() > 1).sum()}\")\n    print(f\"Correlation between original and winsorized scores: {changes['original'].corr(changes['winsorized']):.4f}\")\n    \n    # Top score changes\n    print(\"\\nLargest Score Changes:\")\n    print(\"-\" * 50)\n    print(changes.nlargest(5, 'change')[['feature', 'change']].to_string(index=False))","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:17.198152Z","iopub.execute_input":"2024-10-29T05:47:17.198587Z","iopub.status.idle":"2024-10-29T05:47:17.222421Z","shell.execute_reply.started":"2024-10-29T05:47:17.198547Z","shell.execute_reply":"2024-10-29T05:47:17.221189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create visualization and print stats\nfig = create_comparison_visualizations(evaluation_results)\nprint_summary_stats(evaluation_results)","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:17.223912Z","iopub.execute_input":"2024-10-29T05:47:17.224366Z","iopub.status.idle":"2024-10-29T05:47:18.419317Z","shell.execute_reply.started":"2024-10-29T05:47:17.224325Z","shell.execute_reply":"2024-10-29T05:47:18.417997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\nimport plotly.subplots as sp\nimport plotly.express as px\nimport pandas as pd\nimport numpy as np\nfrom collections import defaultdict\n\ndef create_detailed_analysis(evaluation_results):\n    # Create DataFrames for easier analysis\n    feature_changes = pd.DataFrame([\n        {\n            'feature': k,\n            'change_pct': v['score_change_pct'],\n            'absolute_change_pct': abs(v['score_change_pct']),\n            'change': v['score_change'],\n            'original_score': v['original_score'],\n            'winsorized_score': v['winsorized_score']\n        }\n        for k, v in evaluation_results['comparison'].items()\n    ])\n    \n    # Get top features from both original and winsorized\n    top_features_orig = pd.DataFrame(evaluation_results['original']['rankings'][:10], \n                                   columns=['feature', 'score'])\n    top_features_wins = pd.DataFrame(evaluation_results['winsorized']['rankings'][:10], \n                                   columns=['feature', 'score'])\n    \n    # Get tail predictors\n    tail_predictors = evaluation_results['original']['feature_groups']['tail_predictors']\n    tail_predictor_stats = feature_changes[feature_changes['feature'].isin(tail_predictors)]\n    \n    # Create detailed summary statistics\n    summary_stats = {\n        'Score Changes Overview': {\n            'Mean Absolute Change (%)': feature_changes['absolute_change_pct'].mean(),\n            'Median Absolute Change (%)': feature_changes['absolute_change_pct'].median(),\n            'Max Absolute Change (%)': feature_changes['absolute_change_pct'].max(),\n            'Features with >1% Change': (feature_changes['absolute_change_pct'] > 1).sum(),\n            'Features with >0.5% Change': (feature_changes['absolute_change_pct'] > 0.5).sum(),\n            'Standard Deviation of Changes (%)': feature_changes['change_pct'].std()\n        },\n        'Top 10 Features Comparison': {\n            'Maintained Position': len(set(top_features_orig['feature']) & set(top_features_wins['feature'])),\n            'New Entries': list(set(top_features_wins['feature']) - set(top_features_orig['feature'])),\n            'Dropped Out': list(set(top_features_orig['feature']) - set(top_features_wins['feature']))\n        },\n        'Tail Predictors Analysis': {\n            'Count': len(tail_predictors),\n            'Average Score (Original)': feature_changes[feature_changes['feature'].isin(tail_predictors)]['original_score'].mean(),\n            'Average Score (Winsorized)': feature_changes[feature_changes['feature'].isin(tail_predictors)]['winsorized_score'].mean(),\n            'Average Change (%)': tail_predictor_stats['change_pct'].mean(),\n            'Max Change (%)': tail_predictor_stats['change_pct'].max(),\n            'Min Change (%)': tail_predictor_stats['change_pct'].min()\n        }\n    }\n    \n    # Create visualizations\n    fig = sp.make_subplots(\n        rows=2, cols=2,\n        subplot_titles=(\n            'Original vs Winsorized Scores',\n            'Top 10 Score Changes',\n            'Distribution of Score Changes',\n            'Tail Predictors Score Changes'\n        ),\n        specs=[[{'type': 'scatter'}, {'type': 'bar'}],\n               [{'type': 'histogram'}, {'type': 'bar'}]]\n    )\n    \n    # 1. Scatter plot\n    fig.add_trace(\n        go.Scatter(\n            x=feature_changes['original_score'],\n            y=feature_changes['winsorized_score'],\n            mode='markers',\n            text=feature_changes['feature'],\n            name='Features',\n            marker=dict(\n                size=8,\n                color='blue',\n                opacity=0.6\n            )\n        ),\n        row=1, col=1\n    )\n    \n    # Add diagonal line\n    x_range = [feature_changes['original_score'].min(), feature_changes['original_score'].max()]\n    fig.add_trace(\n        go.Scatter(\n            x=x_range,\n            y=x_range,\n            mode='lines',\n            name='y=x',\n            line=dict(color='red', dash='dash')\n        ),\n        row=1, col=1\n    )\n    \n    # 2. Top changes bar plot\n    top_changes = feature_changes.nlargest(10, 'absolute_change_pct')\n    fig.add_trace(\n        go.Bar(\n            y=top_changes['feature'],\n            x=top_changes['change_pct'],\n            orientation='h',\n            marker_color=['red' if x < 0 else 'green' for x in top_changes['change_pct']],\n            name='Top Changes'\n        ),\n        row=1, col=2\n    )\n    \n    # 3. Distribution histogram\n    fig.add_trace(\n        go.Histogram(\n            x=feature_changes['change_pct'],\n            nbinsx=30,\n            name='Change Distribution'\n        ),\n        row=2, col=1\n    )\n    \n    # 4. Tail predictors changes\n    tail_changes = feature_changes[feature_changes['feature'].isin(tail_predictors)].sort_values('absolute_change_pct', ascending=False)\n    fig.add_trace(\n        go.Bar(\n            y=tail_changes['feature'][:10],\n            x=tail_changes['change_pct'][:10],\n            orientation='h',\n            marker_color=['red' if x < 0 else 'green' for x in tail_changes['change_pct'][:10]],\n            name='Tail Predictor Changes'\n        ),\n        row=2, col=2\n    )\n    \n    # Update layout\n    fig.update_layout(\n        height=900,\n        width=1200,\n        showlegend=True,\n        title_text=\"Feature Score Analysis: Original vs Winsorized\",\n    )\n    \n    # Update axes labels\n    fig.update_xaxes(title_text=\"Original Score\", row=1, col=1)\n    fig.update_yaxes(title_text=\"Winsorized Score\", row=1, col=1)\n    fig.update_xaxes(title_text=\"Score Change (%)\", row=1, col=2)\n    fig.update_xaxes(title_text=\"Score Change (%)\", row=2, col=1)\n    fig.update_yaxes(title_text=\"Count\", row=2, col=1)\n    fig.update_xaxes(title_text=\"Score Change (%)\", row=2, col=2)\n    \n    return fig, summary_stats\n\n# Create the visualization and get summary stats\nfig, summary_stats = create_detailed_analysis(evaluation_results)\n\n# Print detailed summary\nprint(\"\\nDetailed Summary Statistics\")\nprint(\"=\" * 50)\n\nfor category, stats in summary_stats.items():\n    print(f\"\\n{category}:\")\n    print(\"-\" * 40)\n    for metric, value in stats.items():\n        print(f\"{metric}: {value}\")\n\n# Create detailed top features comparison\nprint(\"\\nTop 10 Features Detailed Comparison\")\nprint(\"=\" * 50)\nprint(\"\\nOriginal Top 10:\")\ntop_10_orig = pd.DataFrame(evaluation_results['original']['rankings'][:10])\ntop_10_orig.columns = ['Feature', 'Score']\nprint(top_10_orig.to_string(index=False))\n\nprint(\"\\nWinsorized Top 10:\")\ntop_10_wins = pd.DataFrame(evaluation_results['winsorized']['rankings'][:10])\ntop_10_wins.columns = ['Feature', 'Score']\nprint(top_10_wins.to_string(index=False))\n\n# Show top absolute changes\nprint(\"\\nTop 10 Absolute Score Changes\")\nprint(\"=\" * 50)\nchanges = pd.DataFrame([\n    {\n        'Feature': k,\n        'Original Score': v['original_score'],\n        'Winsorized Score': v['winsorized_score'],\n        'Absolute Change (%)': abs(v['score_change_pct']),\n        'Change Direction': 'Increase' if v['score_change_pct'] > 0 else 'Decrease'\n    }\n    for k, v in evaluation_results['comparison'].items()\n]).sort_values('Absolute Change (%)', ascending=False).head(10)\n\nprint(changes.to_string(index=False))\n\n# Display the Plotly figure\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:18.421240Z","iopub.execute_input":"2024-10-29T05:47:18.421627Z","iopub.status.idle":"2024-10-29T05:47:19.959959Z","shell.execute_reply.started":"2024-10-29T05:47:18.421587Z","shell.execute_reply":"2024-10-29T05:47:19.958550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Winsorized vs Original Score：Scatter Points are significantly ploted along the line, mean both of them are high correlated.Interestingly, there is no any points outlying in the graph, means winsorization cause a general effect to the original data.\n- Top 10 Score Changes: I will treat over 2% change as \"significantly change, means features would have the partical effect to the target variable based on the exterme values. It also means that they are somehow unstable to be used as the baseline of the training, we might need to use feature engineering for these. \n- Distribution of Score Changes: Most data are distributed around (-0.4, 0.2) -> robusted; some others are in (0.8, 3) -> highly sensitive to outliers. Thus we could treat these features into two categories: “stable prediction groups” and “tail risk groups”\n- Tail Predictor Score Changes:Sensitivity of tail predictor features after winsorization. ","metadata":{}},{"cell_type":"markdown","source":"<a id='score_group'></a>\n## Scores by Group","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.graph_objects as go\nimport plotly.express as px\nfrom plotly.subplots import make_subplots\n\ndef analyze_feature_groups(evaluation_results):\n    changes = pd.DataFrame([\n        {\n            'feature': k,\n            'change_pct': v['score_change_pct'],\n            'original_score': v['original_score'],\n            'winsorized_score': v['winsorized_score'],\n            'absolute_change_pct': abs(v['score_change_pct'])\n        }\n        for k, v in evaluation_results['comparison'].items()\n    ])\n    \n    # Top 10\n    top_features = set(evaluation_results['original']['top_features'])\n    \n    # tail predictors\n    tail_predictors = set(evaluation_results['original']['feature_groups']['tail_predictors'])\n    \n    # define features' groups\n    def get_feature_group(row):\n        groups = []\n        if row['feature'] in top_features:\n            groups.append('Top Feature')\n        if row['feature'] in tail_predictors:\n            groups.append('Tail Predictor')\n        \n        # Classifying the range of Change\n        if abs(row['change_pct']) < 0.1:\n            groups.append('Stable')\n        elif row['change_pct'] > 0:\n            groups.append('Positive Change')\n        else:\n            groups.append('Negative Change')\n            \n        return ' & '.join(groups)\n    \n    # labelling\n    changes['group'] = changes.apply(get_feature_group, axis=1)\n    \n    # stat summary\n    stats = {\n        'total_features': len(changes),\n        'top_features': len(top_features),\n        'tail_predictors': len(tail_predictors),\n        'stable_features': len(changes[changes['absolute_change_pct'] < 0.1]),\n        'positive_change': len(changes[changes['change_pct'] > 0.1]),\n        'negative_change': len(changes[changes['change_pct'] < -0.1]),\n        'group_counts': changes['group'].value_counts().to_dict()\n    }\n    \n    # viz\n    # 1. scatter plot Original vs winsorized\n    fig = make_subplots(\n        rows=2, cols=1,\n        subplot_titles=('Feature Scores by Group', 'Score Changes Distribution by Group'),\n    )\n    \n    # add scatters\n    for group in changes['group'].unique():\n        mask = changes['group'] == group\n        fig.add_trace(\n            go.Scatter(\n                x=changes[mask]['original_score'],\n                y=changes[mask]['winsorized_score'],\n                mode='markers',\n                name=group,\n                text=changes[mask]['feature'],\n                hovertemplate=\"Feature: %{text}<br>\" +\n                            \"Original Score: %{x:.4f}<br>\" +\n                            \"Winsorized Score: %{y:.4f}<br>\" +\n                            \"Change: %{customdata:.2f}%\",\n                customdata=changes[mask]['change_pct']\n            ),\n            row=1, col=1\n        )\n    \n    # set y=x axis\n    fig.add_trace(\n        go.Scatter(\n            x=[0, 1],\n            y=[0, 1],\n            mode='lines',\n            name='y=x',\n            line=dict(dash='dash', color='red'),\n            showlegend=True\n        ),\n        row=1, col=1\n    )\n    \n    # 2. box plot\n    for group in changes['group'].unique():\n        mask = changes['group'] == group\n        fig.add_trace(\n            go.Box(\n                y=changes[mask]['change_pct'],\n                name=group,\n                boxpoints='all',\n                jitter=0.3,\n                pointpos=-1.8\n            ),\n            row=2, col=1\n        )\n    \n    fig.update_layout(\n        height=1000,\n        title_text=\"Feature Analysis by Groups\",\n        showlegend=True\n    )\n    \n    fig.update_xaxes(title_text=\"Original Score\", row=1, col=1)\n    fig.update_yaxes(title_text=\"Winsorized Score\", row=1, col=1)\n    fig.update_xaxes(title_text=\"Group\", row=2, col=1)\n    fig.update_yaxes(title_text=\"Score Change (%)\", row=2, col=1)\n    \n    return fig, stats, changes\n\nfig, stats, changes_df = analyze_feature_groups(evaluation_results)\n\nprint(\"\\nFeature Group Analysis\")\nprint(\"=\" * 50)\nprint(f\"\\nOverall Statistics:\")\nprint(f\"Total Features: {stats['total_features']}\")\nprint(f\"Top Features: {stats['top_features']}\")\nprint(f\"Tail Predictors: {stats['tail_predictors']}\")\nprint(f\"Stable Features (change < 0.1%): {stats['stable_features']}\")\nprint(f\"Features with Positive Change (> 0.1%): {stats['positive_change']}\")\nprint(f\"Features with Negative Change (< -0.1%): {stats['negative_change']}\")\n\nprint(\"\\nDetailed Group Counts:\")\nfor group, count in stats['group_counts'].items():\n    print(f\"{group}: {count}\")\n\nprint(\"\\nFeatures in Each Group:\")\nfor group in changes_df['group'].unique():\n    features = changes_df[changes_df['group'] == group]['feature'].tolist()\n    print(f\"\\n{group}:\")\n    print(\", \".join(features))\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:19.961859Z","iopub.execute_input":"2024-10-29T05:47:19.962226Z","iopub.status.idle":"2024-10-29T05:47:20.097817Z","shell.execute_reply.started":"2024-10-29T05:47:19.962191Z","shell.execute_reply":"2024-10-29T05:47:20.096594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\n\ndef create_binned_analysis(evaluation_results):\n    # DataFrame\n    changes = pd.DataFrame([\n        {\n            'feature': k,\n            'change_pct': v['score_change_pct'],\n            'original_score': v['original_score'],\n            'winsorized_score': v['winsorized_score'],\n            'absolute_change_pct': abs(v['score_change_pct'])\n        }\n        for k, v in evaluation_results['comparison'].items()\n    ])\n    \n    # Top 10 & tail predictors\n    top_features = set(evaluation_results['original']['top_features'])\n    tail_predictors = set(evaluation_results['original']['feature_groups']['tail_predictors'])\n    \n    # labelling\n    changes['feature_type'] = changes['feature'].apply(\n        lambda x: 'Top 10 & Tail' if x in top_features and x in tail_predictors\n        else 'Top 10' if x in top_features\n        else 'Tail Predictor' if x in tail_predictors\n        else 'Other'\n    )\n    \n    # create bin\n    bins = np.linspace(0.1, 0.6, 11)  # ten equal bin intervals\n    changes['score_bin'] = pd.cut(changes['original_score'], \n                                bins=bins, \n                                labels=[f'{bins[i]:.2f}-{bins[i+1]:.2f}' for i in range(len(bins)-1)])\n    \n    color_map = {\n        'Top 10 & Tail': '#FF5733',  \n        'Top 10': '#33FF57',         \n        'Tail Predictor': '#3357FF', \n        'Other': '#808080'\n    }\n    \n    # bubble plot\n    fig = make_subplots(\n        rows=2, cols=1,\n        subplot_titles=('Feature Distribution by Score Ranges', 'Change Distribution by Feature Type'),\n        vertical_spacing=0.12\n    )\n    \n    # bubble for each categories\n    for feature_type in color_map.keys():\n        mask = changes['feature_type'] == feature_type\n        df_group = changes[mask]\n        \n        # count features in each bin\n        bin_counts = df_group.groupby('score_bin').size()\n        \n        fig.add_trace(\n            go.Scatter(\n                x=bin_counts.index,\n                y=[feature_type] * len(bin_counts),\n                mode='markers',\n                name=feature_type,\n                marker=dict(\n                    size=bin_counts.values * 5,  # set bubble size\n                    color=color_map[feature_type],\n                    sizemode='area',\n                    sizeref=2.*max(bin_counts.values)/(40.**2),\n                    sizemin=4\n                ),\n                text=[f\"{bin_counts.index[i]}<br>Count: {bin_counts.values[i]}<br>Features: {', '.join(df_group[df_group['score_bin'] == bin_counts.index[i]]['feature'])}\" \n                      for i in range(len(bin_counts))],\n                hovertemplate=\"Bin: %{text}<extra></extra>\"\n            ),\n            row=1, col=1\n        )\n    \n    # box plot for change dist\n    for feature_type in color_map.keys():\n        mask = changes['feature_type'] == feature_type\n        fig.add_trace(\n            go.Box(\n                y=changes[mask]['change_pct'],\n                name=feature_type,\n                marker_color=color_map[feature_type],\n                boxpoints='all',\n                jitter=0.3,\n                pointpos=-1.8,\n                hoveron=\"points+boxes\",\n                hovertemplate=\"Feature: %{text}<br>Change: %{y:.3f}%<extra></extra>\",\n                text=changes[mask]['feature']\n            ),\n            row=2, col=1\n        )\n    \n    fig.update_layout(\n        height=1000,\n        title_text=\"Feature Analysis with Score Binning\",\n        showlegend=True,\n        legend=dict(\n            orientation=\"h\",\n            yanchor=\"bottom\",\n            y=1.02,\n            xanchor=\"right\",\n            x=1\n        )\n    )\n    \n    fig.update_yaxes(title_text=\"Feature Type\", row=1, col=1)\n    fig.update_xaxes(title_text=\"Score Range\", row=1, col=1)\n    fig.update_xaxes(tickangle=45)\n    \n    fig.update_yaxes(title_text=\"Score Change (%)\", row=2, col=1)\n    fig.update_xaxes(title_text=\"Feature Type\", row=2, col=1)\n    \n    print(\"\\nFeature Distribution Analysis\")\n    print(\"=\" * 50)\n    \n    # Summarize by score range\n    print(\"\\nFeature Counts by Score Range and Type:\")\n    pivot_table = pd.pivot_table(\n        changes, \n        values='feature',\n        index='score_bin',\n        columns='feature_type',\n        aggfunc='count',\n        fill_value=0\n    )\n    print(pivot_table)\n\n    print(\"\\nFeature Type Summary:\")\n    type_summary = changes['feature_type'].value_counts()\n    for ftype, count in type_summary.items():\n        print(f\"{ftype}: {count}\")\n    \n    # Score Range\n    print(\"\\nScore Range Summary:\")\n    bin_summary = changes['score_bin'].value_counts().sort_index()\n    for bin_range, count in bin_summary.items():\n        features_in_bin = changes[changes['score_bin'] == bin_range]['feature'].tolist()\n        print(f\"\\n{bin_range}:\")\n        print(f\"Count: {count}\")\n        print(f\"Features: {', '.join(features_in_bin)}\")\n    \n    return fig, changes\n\nfig, changes_df = create_binned_analysis(evaluation_results)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:20.099581Z","iopub.execute_input":"2024-10-29T05:47:20.099943Z","iopub.status.idle":"2024-10-29T05:47:20.282176Z","shell.execute_reply.started":"2024-10-29T05:47:20.099908Z","shell.execute_reply":"2024-10-29T05:47:20.281181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='corr_features_features'></a>\n# Correlation Features vs Features","metadata":{}},{"cell_type":"code","source":"# init correlation matrix\ncorr_matrix = np.ones((79,79), dtype='float64')\n\n# calc correlations pairwise\nj = 0 # target index\nfor t in features:\n    print('Calculating for target', t)\n    i = 0 # feature index\n    for f in features:\n        corr_ft = df.select(pl.corr(f, t, method='pearson'))[0,0]\n        corr_matrix[i][j] = corr_ft\n        i = i + 1\n    j = j + 1","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:20.283633Z","iopub.execute_input":"2024-10-29T05:47:20.284005Z","iopub.status.idle":"2024-10-29T05:47:20.572649Z","shell.execute_reply.started":"2024-10-29T05:47:20.283962Z","shell.execute_reply":"2024-10-29T05:47:20.571310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert to data frame\ncorr_features_features = pd.DataFrame(corr_matrix, columns=features)\ncorr_features_features.index = features","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:20.574208Z","iopub.execute_input":"2024-10-29T05:47:20.574605Z","iopub.status.idle":"2024-10-29T05:47:21.766000Z","shell.execute_reply.started":"2024-10-29T05:47:20.574564Z","shell.execute_reply":"2024-10-29T05:47:21.764191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show correlations between features and targets\ncorr_features_features","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.767088Z","iopub.status.idle":"2024-10-29T05:47:21.767564Z","shell.execute_reply.started":"2024-10-29T05:47:21.767342Z","shell.execute_reply":"2024-10-29T05:47:21.767365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize correlations\nplt.figure(figsize=(12,12))\nsns.heatmap(corr_features_features, cmap='RdYlGn',\n            linewidths=0.5, linecolor='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.769569Z","iopub.status.idle":"2024-10-29T05:47:21.770028Z","shell.execute_reply.started":"2024-10-29T05:47:21.769819Z","shell.execute_reply":"2024-10-29T05:47:21.769842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\n\n# Define features to highlight\nhighlighted_features = [\n    'feature_08', 'feature_44', 'feature_36', 'feature_04', 'feature_20',\n    'feature_41', 'feature_43', 'feature_33', 'feature_61', 'feature_40'\n]\n\n# Set up the plot\nplt.figure(figsize=(12, 12))\n\n# Create a mask that highlights specific rows\nmask = np.zeros_like(corr_features_features, dtype=bool)\nmask[~corr_features_features.index.isin(highlighted_features), :] = True\n\n# Plot the heatmap, setting mask and custom color for highlighted rows\nsns.heatmap(corr_features_features, cmap='RdYlGn', linewidths=0.5, linecolor='black',\n            mask=mask, cbar_kws={\"shrink\": .8}, fmt=\".2f\")\n\n# Overlay another heatmap to make highlighted rows stand out with a strong color\nsns.heatmap(corr_features_features, mask=~mask, cmap=\"coolwarm\", cbar=False,\n            linewidths=0.8, linecolor='black', alpha=0.75)\n\n# Display plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.771549Z","iopub.status.idle":"2024-10-29T05:47:21.772034Z","shell.execute_reply.started":"2024-10-29T05:47:21.771839Z","shell.execute_reply":"2024-10-29T05:47:21.771861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id='corr_target_features'></a>\n# Correlation Targets vs Features","metadata":{}},{"cell_type":"code","source":"# init correlation matrix\ncorr_matrix = np.ones((79,9), dtype='float64')\n\n# calc correlations pairwise\nj = 0 # target index\nfor t in targets:\n    print('Calculating for target', t)\n    i = 0 # feature index\n    for f in features:\n        corr_ft = df.select(pl.corr(f, t, method='pearson'))[0,0]\n        corr_matrix[i][j] = corr_ft\n        i = i + 1\n    j = j + 1","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.774900Z","iopub.status.idle":"2024-10-29T05:47:21.775358Z","shell.execute_reply.started":"2024-10-29T05:47:21.775140Z","shell.execute_reply":"2024-10-29T05:47:21.775168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert to data frame\ncorr_targets_features = pd.DataFrame(corr_matrix, columns=targets)\ncorr_targets_features.index = features","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.776728Z","iopub.status.idle":"2024-10-29T05:47:21.777201Z","shell.execute_reply.started":"2024-10-29T05:47:21.776970Z","shell.execute_reply":"2024-10-29T05:47:21.776990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show correlations between features and targets\ncorr_targets_features","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.778524Z","iopub.status.idle":"2024-10-29T05:47:21.778936Z","shell.execute_reply.started":"2024-10-29T05:47:21.778721Z","shell.execute_reply":"2024-10-29T05:47:21.778740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize correlations\nplt.figure(figsize=(6,16))\nsns.heatmap(corr_targets_features, cmap='RdYlGn',\n            linewidths=0.5, linecolor='black')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.782830Z","iopub.status.idle":"2024-10-29T05:47:21.784032Z","shell.execute_reply.started":"2024-10-29T05:47:21.783761Z","shell.execute_reply":"2024-10-29T05:47:21.783798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# export result\ncorr_targets_features.to_csv('corr_targets_features.csv')","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.785320Z","iopub.status.idle":"2024-10-29T05:47:21.785779Z","shell.execute_reply.started":"2024-10-29T05:47:21.785561Z","shell.execute_reply":"2024-10-29T05:47:21.785583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualize an example pair\nf = 'feature_08'\nt = 'responder_6'\nsns.jointplot(data=df, x=f, y=t,\n              kind='hist')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-29T05:47:21.787343Z","iopub.status.idle":"2024-10-29T05:47:21.787743Z","shell.execute_reply.started":"2024-10-29T05:47:21.787543Z","shell.execute_reply":"2024-10-29T05:47:21.787563Z"},"trusted":true},"execution_count":null,"outputs":[]}]}