{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split, cross_val_score, StratifiedKFold, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import classification_report, roc_auc_score, confusion_matrix, accuracy_score\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# ------------------------------------------------------------\n# 1. Load data\n# ------------------------------------------------------------\nfile_path = \"/kaggle/input/datasets/rustam32/real-airline-passenger-satisfaction-dataset/passenger_survey_balanced.csv\"\ndf = pd.read_csv(file_path, low_memory=False)\nprint(f\"Dataset shape: {df.shape}\")\n\n# ------------------------------------------------------------\n# 2. Target column\n# ------------------------------------------------------------\ntarget = 'liked'\nif target not in df.columns:\n    raise KeyError(f\"Target '{target}' not found. Columns: {df.columns.tolist()[:10]}...\")\n\n# ------------------------------------------------------------\n# 3. Handle applicability flags\n# ------------------------------------------------------------\nrating_cols = [col for col in df.columns if not col.endswith('_is_applicable') \n               and col not in [target, 'process', 'month', 'flight_type', 'connection',\n                               'ticket_purchased_by', 'ticket_purchase_channel', 'transport_to_airport',\n                               'has_disability', 'uses_assistive_device', 'requested_special_assistance',\n                               'disembarkation_method_used', 'nationality', 'gender', 'age_group',\n                               'education', 'household_income', 'traveling_alone', 'number_of_companions',\n                               'trip_purpose', 'trips_last_12_months', 'used_airport_before_last_12_months',\n                               'arrival_lead_time', 'connection_wait_time']]\n\napplicable_cols = [col for col in df.columns if col.endswith('_is_applicable')]\n\nfor rating in rating_cols:\n    flag = rating + '_is_applicable'\n    if flag in df.columns:\n        df.loc[df[flag] == 0, rating] = np.nan\n\ndf = df.drop(columns=applicable_cols)\n\n# ------------------------------------------------------------\n# 4. Clean rating columns\n# ------------------------------------------------------------\nfor col in rating_cols:\n    if col in df.columns:\n        df[col] = pd.to_numeric(df[col], errors='coerce')\n\nnumeric_features = [col for col in rating_cols if col in df.columns and df[col].notna().any()]\n\n# ------------------------------------------------------------\n# 5. Separate features\n# ------------------------------------------------------------\nX_numeric = df[numeric_features].astype(float)\n\ncategorical_features = ['process', 'month', 'flight_type', 'connection', 'ticket_purchased_by',\n                        'ticket_purchase_channel', 'transport_to_airport', 'has_disability',\n                        'nationality', 'gender', 'age_group', 'education', 'household_income',\n                        'traveling_alone', 'trip_purpose', 'trips_last_12_months',\n                        'used_airport_before_last_12_months']\n\nfor col in categorical_features:\n    if col in df.columns:\n        df[col] = df[col].fillna('Unknown').astype(str)\n\nX_cat = df[categorical_features]\ny = df[target].astype(int)\n\nvalid_idx = y.notna()\nX_numeric = X_numeric[valid_idx]\nX_cat = X_cat[valid_idx]\ny = y[valid_idx]\n\nprint(f\"Final rows: {len(y)}\")\nprint(f\"Target distribution:\\n{y.value_counts()}\")\n\n# ------------------------------------------------------------\n# 6. Preprocessing pipeline \n# ------------------------------------------------------------\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n    ('onehot', OneHotEncoder(handle_unknown='ignore', sparse_output=False))\n])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ])\n\n# ------------------------------------------------------------\n# 7. Train / test split\n# ------------------------------------------------------------\nX = pd.concat([X_numeric, X_cat], axis=1)\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# ------------------------------------------------------------\n# 8. Define models to compare\n# ------------------------------------------------------------\nmodels = {\n    'Random Forest': RandomForestClassifier(n_estimators=100, max_depth=10, random_state=42, n_jobs=-1),\n    'Gradient Boosting': GradientBoostingClassifier(n_estimators=100, max_depth=5, random_state=42),\n    'Logistic Regression': LogisticRegression(max_iter=1000, random_state=42)\n}\n\nif xgboost_available:\n    models['XGBoost'] = XGBClassifier(n_estimators=100, max_depth=5, random_state=42, use_label_encoder=False, eval_metric='logloss')\nif lightgbm_available:\n    models['LightGBM'] = LGBMClassifier(n_estimators=100, max_depth=5, random_state=42, verbose=-1)\n\n# ------------------------------------------------------------\n# 9. Train and evaluate all models\n# ------------------------------------------------------------\nresults = []\nbest_model = None\nbest_accuracy = 0\n\nprint(\"\\n\" + \"=\"*60)\nprint(\"MODEL COMPARISON\")\nprint(\"=\"*60)\n\nfor name, clf in models.items():\n    print(f\"\\nTraining {name}...\")\n    pipeline = Pipeline(steps=[('preprocessor', preprocessor), ('classifier', clf)])\n    pipeline.fit(X_train, y_train)\n    y_pred = pipeline.predict(X_test)\n    y_proba = pipeline.predict_proba(X_test)[:, 1]\n    \n    acc = accuracy_score(y_test, y_pred)\n    roc = roc_auc_score(y_test, y_proba)\n    results.append({'Model': name, 'Accuracy': acc, 'ROC-AUC': roc})\n    \n    print(f\"{name} - Accuracy: {acc:.4f}, ROC-AUC: {roc:.4f}\")\n    \n    if acc > best_accuracy:\n        best_accuracy = acc\n        best_model = pipeline\n\n# Display comparison table\nresults_df = pd.DataFrame(results).sort_values('Accuracy', ascending=False)\nprint(\"\\n\" + \"=\"*60)\nprint(\"SUMMARY\")\nprint(results_df.to_string(index=False))\nprint(\"=\"*60)\n\n# ------------------------------------------------------------\n# 10. Hyperparameter tuning for the best model\n# ------------------------------------------------------------\nprint(\"\\n\" + \"=\"*60)\nprint(\"HYPERPARAMETER TUNING FOR BEST MODEL\")\nprint(\"=\"*60)\n\n# Identify the best base model name\nbest_model_name = results_df.iloc[0]['Model']\nprint(f\"Tuning {best_model_name}...\")\n\n# Define parameter grids based on model type\nif best_model_name == 'Random Forest':\n    param_grid = {\n        'classifier__n_estimators': [100, 200],\n        'classifier__max_depth': [10, 15, None],\n        'classifier__min_samples_split': [2, 5]\n    }\nelif best_model_name == 'Gradient Boosting':\n    param_grid = {\n        'classifier__n_estimators': [100, 200],\n        'classifier__max_depth': [3, 5, 7],\n        'classifier__learning_rate': [0.05, 0.1]\n    }\nelif best_model_name == 'XGBoost':\n    param_grid = {\n        'classifier__n_estimators': [100, 200],\n        'classifier__max_depth': [3, 5, 7],\n        'classifier__learning_rate': [0.05, 0.1]\n    }\nelif best_model_name == 'LightGBM':\n    param_grid = {\n        'classifier__n_estimators': [100, 200],\n        'classifier__max_depth': [3, 5, 7],\n        'classifier__learning_rate': [0.05, 0.1]\n    }\nelse:  # Logistic Regression\n    param_grid = {\n        'classifier__C': [0.1, 1.0, 10.0],\n        'classifier__penalty': ['l2']\n    }\n\n# Create a pipeline with the best model \nif best_model_name == 'Random Forest':\n    base_clf = RandomForestClassifier(random_state=42, n_jobs=-1)\nelif best_model_name == 'Gradient Boosting':\n    base_clf = GradientBoostingClassifier(random_state=42)\nelif best_model_name == 'XGBoost' and xgboost_available:\n    base_clf = XGBClassifier(random_state=42, use_label_encoder=False, eval_metric='logloss')\nelif best_model_name == 'LightGBM' and lightgbm_available:\n    base_clf = LGBMClassifier(random_state=42, verbose=-1)\nelse:\n    base_clf = LogisticRegression(max_iter=1000, random_state=42)\n\npipeline_tune = Pipeline(steps=[('preprocessor', preprocessor), ('classifier', base_clf)])\n\ngrid_search = GridSearchCV(pipeline_tune, param_grid, cv=3, scoring='accuracy', n_jobs=-1, verbose=1)\ngrid_search.fit(X_train, y_train)\n\nprint(f\"Best parameters: {grid_search.best_params_}\")\nprint(f\"Best cross-validation accuracy: {grid_search.best_score_:.4f}\")\n\n# Evaluate tuned model on test set\ny_pred_tuned = grid_search.predict(X_test)\ny_proba_tuned = grid_search.predict_proba(X_test)[:, 1]\nacc_tuned = accuracy_score(y_test, y_pred_tuned)\nroc_tuned = roc_auc_score(y_test, y_proba_tuned)\n\nprint(f\"Tuned model test accuracy: {acc_tuned:.4f}\")\nprint(f\"Tuned model test ROC-AUC: {roc_tuned:.4f}\")\n\n# ------------------------------------------------------------\n# 11. Feature importance \n# ------------------------------------------------------------\nif best_model_name in ['Random Forest', 'Gradient Boosting', 'XGBoost', 'LightGBM']:\n    # Get feature names after one-hot encoding\n    cat_feature_names = preprocessor.named_transformers_['cat'].named_steps['onehot'].get_feature_names_out(categorical_features)\n    all_feature_names = numeric_features + list(cat_feature_names)\n    \n    best_clf = grid_search.best_estimator_.named_steps['classifier']\n    importances = best_clf.feature_importances_\n    indices = np.argsort(importances)[::-1][:20]\n    \n    plt.figure(figsize=(10, 6))\n    plt.title(f\"Top 20 Feature Importances - {best_model_name} (Tuned)\")\n    plt.barh(range(len(indices)), importances[indices], align='center')\n    plt.yticks(range(len(indices)), [all_feature_names[i] for i in indices])\n    plt.gca().invert_yaxis()\n    plt.xlabel(\"Relative Importance\")\n    plt.tight_layout()\n    plt.show()\nelse:\n    print(\"Feature importance plot is only available for tree-based models.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-22T10:44:10.794646Z","iopub.execute_input":"2026-05-22T10:44:10.794974Z","iopub.status.idle":"2026-05-22T10:58:03.989949Z","shell.execute_reply.started":"2026-05-22T10:44:10.794946Z","shell.execute_reply":"2026-05-22T10:58:03.988869Z"}},"outputs":[],"execution_count":null}]}