{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":7163,"databundleVersionId":44582}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- CELL 1: IMPORTS ---\nimport pandas as pd\nimport numpy as np\nimport xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, roc_auc_score\nimport gc\n\n# Define Kaggle's default data path\nDATA_PATH = \"/kaggle/input/competitions/kkbox-churn-prediction-challenge/\"\nprint(\"Environment setup complete. Ready for data.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:24:13.636924Z","iopub.execute_input":"2026-05-07T11:24:13.637237Z","iopub.status.idle":"2026-05-07T11:24:14.841738Z","shell.execute_reply.started":"2026-05-07T11:24:13.637211Z","shell.execute_reply":"2026-05-07T11:24:14.840807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 1: EXTRACTING THE REQUIRED DATA ---\nimport os\n\n# Create a folder to hold our unzipped data so our workspace stays clean\nOUTPUT_DIR = \"/kaggle/working/data/churn_comp_refresh/\"\nos.makedirs(OUTPUT_DIR, exist_ok=True)\n\n# The path where Kaggle stores the downloaded competition files\nKAGGLE_INPUT_DIR = \"/kaggle/input/competitions/kkbox-churn-prediction-challenge/\"\n\n\nprint(\"1. Extracting train_v2.csv...\")\n!7za e {KAGGLE_INPUT_DIR}train_v2.csv.7z -o{OUTPUT_DIR} -y\n\nprint(\"\\n2. Extracting transactions_v2.csv (This might take a minute)...\")\n!7za e {KAGGLE_INPUT_DIR}transactions_v2.csv.7z -o{OUTPUT_DIR} -y\n\nprint(\"\\n--- EXTRACTION COMPLETE ---\")\nprint(\"Files ready in your working directory:\")\nprint(os.listdir(OUTPUT_DIR))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:29:04.302021Z","iopub.execute_input":"2026-05-07T11:29:04.302405Z","iopub.status.idle":"2026-05-07T11:29:09.689033Z","shell.execute_reply.started":"2026-05-07T11:29:04.302365Z","shell.execute_reply":"2026-05-07T11:29:09.688195Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install -q py7zr\n\n# import os\n# import py7zr\n\n# DATA_PATH = \"/kaggle/input/competitions/kkbox-churn-prediction-challenge/\"\n# EXTRACT_PATH = \"/kaggle/working/\"\n\n# # Get all .7z files\n# archives = [f for f in os.listdir(DATA_PATH) if f.endswith(\".7z\")]\n\n# print(\"Archives found:\")\n# for a in archives:\n#     print(a)\n\n# # Extract all archives\n# for archive in archives:\n#     archive_path = os.path.join(DATA_PATH, archive)\n\n#     print(f\"\\nExtracting: {archive}\")\n\n#     with py7zr.SevenZipFile(archive_path, mode='r') as z:\n#         z.extractall(path=EXTRACT_PATH)\n\n# print(\"\\nExtraction complete!\")\n\n# # Explore extracted folders\n# print(\"\\nWorking directory:\")\n# print(os.listdir(\"/kaggle/working\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:25:13.073176Z","iopub.execute_input":"2026-05-07T11:25:13.073462Z","iopub.status.idle":"2026-05-07T11:25:25.454631Z","shell.execute_reply.started":"2026-05-07T11:25:13.073434Z","shell.execute_reply":"2026-05-07T11:25:25.453167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 2: LOAD & DOWNSAMPLE DATA ---\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport gc\n\n# The path where we just extracted our files\nDATA_PATH = '/kaggle/working/data/churn_comp_refresh/'\n\nprint(\"1. Loading training labels (v2)...\")\ntrain_raw = pd.read_csv(DATA_PATH + 'train_v2.csv')\n\nprint(\"2. Downsampling to 20,000 users for our MVP...\")\n# Stratified sampling ensures our MVP sees the realistic ratio of churners to non-churners\ntrain_sample, _ = train_test_split(\n    train_raw, \n    train_size=20000, \n    stratify=train_raw['is_churn'], \n    random_state=42\n)\n\n# We convert the IDs to a Python 'set'. Searching a set is O(1) (instant), \n# which makes filtering the massive transactions file much faster.\nsampled_users = set(train_sample['msno'])\nprint(f\"Sampled {len(sampled_users)} unique users.\")\n\nprint(\"3. Loading transactions (v2) for our sampled users...\")\n# We read the transactions file in chunks of 500,000 rows to save RAM\nchunk_iterator = pd.read_csv(DATA_PATH + 'transactions_v2.csv', chunksize=500000)\nfiltered_chunks = []\n\nfor i, chunk in enumerate(chunk_iterator):\n    # Keep only the rows belonging to our 20,000 sampled users\n    filtered_chunk = chunk[chunk['msno'].isin(sampled_users)]\n    filtered_chunks.append(filtered_chunk)\n    print(f\"  - Processed chunk {i+1}...\")\n\ntransactions_df = pd.concat(filtered_chunks, ignore_index=True)\nprint(f\"\\nExtraction complete! We have {len(transactions_df)} transaction records for our 20k users.\")\n\n# Clean up RAM immediately (Crucial for production stability)\ndel train_raw\ngc.collect()\nprint(\"Memory cleaned.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:29:48.956695Z","iopub.execute_input":"2026-05-07T11:29:48.956958Z","iopub.status.idle":"2026-05-07T11:29:51.785912Z","shell.execute_reply.started":"2026-05-07T11:29:48.956933Z","shell.execute_reply":"2026-05-07T11:29:51.785022Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 3: FEATURE ENGINEERING (RFM) ---\nprint(\"1. Converting dates to datetime objects...\")\n# Convert integer dates (e.g., 20170228) to proper Pandas datetime objects\ntransactions_df['transaction_date'] = pd.to_datetime(transactions_df['transaction_date'], format='%Y%m%d')\ntransactions_df['membership_expire_date'] = pd.to_datetime(transactions_df['membership_expire_date'], format='%Y%m%d')\n\nprint(\"2. Engineering RFM Features...\")\n# Group by user ID ('msno') and calculate aggregate statistics\nfeatures_df = transactions_df.groupby('msno').agg(\n    total_transactions=('payment_method_id', 'count'),\n    total_cancellations=('is_cancel', 'sum'),\n    auto_renew_count=('is_auto_renew', 'sum'),\n    total_amount_paid=('actual_amount_paid', 'sum'),\n    avg_plan_price=('plan_list_price', 'mean'),\n    first_transaction_date=('transaction_date', 'min'),\n    last_transaction_date=('transaction_date', 'max')\n).reset_index()\n\n# Calculate custom Recency & Behavioral metrics\nfeatures_df['billing_tenure_days'] = (features_df['last_transaction_date'] - features_df['first_transaction_date']).dt.days\nfeatures_df['cancel_rate'] = features_df['total_cancellations'] / features_df['total_transactions']\n\n# Drop datetime columns because the ML model only understands numbers\nfeatures_df.drop(columns=['first_transaction_date', 'last_transaction_date'], inplace=True)\n\nprint(f\"Feature Engineering complete! We now have features for {len(features_df)} unique users.\")\nprint(\"\\nPreview of the Feature Matrix:\")\ndisplay(features_df.head(5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:29:53.985656Z","iopub.execute_input":"2026-05-07T11:29:53.985932Z","iopub.status.idle":"2026-05-07T11:29:54.092880Z","shell.execute_reply.started":"2026-05-07T11:29:53.985906Z","shell.execute_reply":"2026-05-07T11:29:54.091970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 4: THE MERGE ---\nprint(\"1. Merging labels with our engineered features...\")\n# We use a 'left' merge. This keeps ALL 20,000 users from train_sample.\n# If they don't exist in features_df, Pandas will fill their feature columns with NaN.\nfinal_matrix = pd.merge(train_sample, features_df, on='msno', how='left')\n\nprint(\"2. Handling users with no transaction history...\")\n# We fill those NaNs with 0. \n# 0 transactions, 0 dollars paid, 0 cancellations.\nfinal_matrix.fillna(0, inplace=True)\n\nprint(f\"Final Matrix Shape: {final_matrix.shape} (Notice it's back to 20,0000!)\")\ndisplay(final_matrix.head(3))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:29:59.050544Z","iopub.execute_input":"2026-05-07T11:29:59.050839Z","iopub.status.idle":"2026-05-07T11:29:59.094458Z","shell.execute_reply.started":"2026-05-07T11:29:59.050812Z","shell.execute_reply":"2026-05-07T11:29:59.093770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- END-TO-END AUTO-TUNING & TRAINING PIPELINE ---\nimport xgboost as xgb\nfrom sklearn.metrics import classification_report, roc_auc_score\nfrom sklearn.model_selection import train_test_split, RandomizedSearchCV\nimport warnings\nwarnings.filterwarnings('ignore') # Hides annoying deprecation warnings from sklearn\n\nprint(\"1. Preparing Data Splits...\")\nX = final_matrix.drop(columns=['msno', 'is_churn'])\ny = final_matrix['is_churn']\n\n# SPLIT 1: The \"Final Exam\" (20%)\nX_temp, X_test, y_temp, y_test = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# SPLIT 2: \"Study Guide\" (60%) and \"Practice Quiz\" (20%)\nX_train, X_val, y_train, y_val = train_test_split(\n    X_temp, y_temp, test_size=0.25, random_state=42, stratify=y_temp\n)\n\nprint(\"\\n2. Defining the Hyperparameter Search Space...\")\n# These are the mathematical gears we want the AI to test\nparam_grid = {\n    'max_depth': [3, 4, 5, 6],                  # Tree complexity\n    'learning_rate': [0.01, 0.05, 0.1, 0.2],    # Speed of learning\n    'scale_pos_weight': [5, 7, 10, 12],         # Precision vs Recall dial\n    'subsample': [0.7, 0.8, 0.9, 1.0],          # % of rows used per tree\n    'colsample_bytree': [0.7, 0.8, 0.9, 1.0]    # % of columns used per tree\n}\n\nprint(\"3. Running Randomized Search (Hunting for best parameters)...\")\n# We use a basic model to test the gears quickly\ntune_model = xgb.XGBClassifier(n_estimators=100, random_state=42, eval_metric=\"logloss\")\n\nrandom_search = RandomizedSearchCV(\n    estimator=tune_model,\n    param_distributions=param_grid,\n    n_iter=15,             # Tests 15 random combinations (Increase to 50 for a deeper search!)\n    scoring='f1',          # Optimize for the F1-Score\n    cv=3,                  # 3-Fold Cross Validation\n    verbose=1,             \n    random_state=42,\n    n_jobs=-1              # Use all CPU cores to speed it up\n)\n\n# Fit the search engine on the training data\nrandom_search.fit(X_train, y_train)\n\nbest_params = random_search.best_params_\nprint(f\"\\n✅ Hunt Complete! Best Parameters Found: \\n{best_params}\")\n\nprint(\"\\n4. Training Final Production Model with Early Stopping...\")\n# We initialize a brand new model using the **best_params discovered above\nfinal_model = xgb.XGBClassifier(\n    **best_params,             \n    n_estimators=1000,         # High limit, we rely on early stopping\n    random_state=42,\n    early_stopping_rounds=15,  # Stop if it stops improving\n    eval_metric=\"auc\"          \n)\n\n# Train the final model, using the Validation set to monitor for early stopping\nfinal_model.fit(\n    X_train, y_train,\n    eval_set=[(X_train, y_train), (X_val, y_val)],\n    verbose=50  \n)\n\nprint(f\"\\nModel stopped automatically at tree #{final_model.best_iteration}\")\n\nprint(\"\\n5. Evaluating Final Model on UNSEEN Test Data...\")\ny_probs = final_model.predict_proba(X_test)[:, 1]\ny_pred = (y_probs >= 0.4).astype(int)\n\nprint(\"\\n--- ULTIMATE CLASSIFICATION REPORT ---\")\nprint(classification_report(y_test, y_pred))\nprint(f\"ROC-AUC Score: {roc_auc_score(y_test, y_probs):.3f}\")\n\nprint(\"\\n6. Saving the highly-optimized model...\")\nfinal_model.save_model(\"ai_retention_xgboost_optimized.json\")\nprint(\"Model saved successfully as 'ai_retention_xgboost_optimized.json'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T11:30:00.346694Z","iopub.execute_input":"2026-05-07T11:30:00.346978Z","iopub.status.idle":"2026-05-07T11:30:03.639941Z","shell.execute_reply.started":"2026-05-07T11:30:00.346953Z","shell.execute_reply":"2026-05-07T11:30:03.639240Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 6: THRESHOLD OPTIMIZATION ---\nfrom sklearn.metrics import precision_recall_curve\nimport numpy as np\n\nprint(\"1. Hunting for the Optimal Threshold on the Validation Set...\")\n# We use the Validation set to find the threshold so we don't cheat on the Test set\ny_val_probs = final_model.predict_proba(X_val)[:, 1]\n\n# precision_recall_curve tests hundreds of thresholds automatically\nprecisions, recalls, thresholds = precision_recall_curve(y_val, y_val_probs)\n\n# Calculate the F1 Score for every single threshold\n# (Adding 1e-9 prevents division by zero errors)\nf1_scores = 2 * (precisions * recalls) / (precisions + recalls + 1e-9)\n\n# Find the index of the highest F1 Score\nbest_index = np.argmax(f1_scores)\noptimal_threshold = thresholds[best_index]\nbest_val_f1 = f1_scores[best_index]\n\nprint(f\"✅ Hunt Complete!\")\nprint(f\"The default threshold is 0.500\")\nprint(f\"The mathematical OPTIMAL threshold is: {optimal_threshold:.3f}\")\nprint(f\"This threshold yields a Validation F1-Score of: {best_val_f1:.3f}\\n\")\n\n\nprint(\"2. Applying the Optimal Threshold to the UNSEEN Test Data...\")\n# Get the raw probabilities for the Test set\ny_test_probs = final_model.predict_proba(X_test)[:, 1]\n\n# Convert to hard predictions using our NEW optimal threshold instead of 0.5\ny_pred_optimized = (y_test_probs >= optimal_threshold).astype(int)\n\nprint(\"\\n--- OPTIMIZED PRODUCTION CLASSIFICATION REPORT ---\")\nprint(classification_report(y_test, y_pred_optimized))\n\n# Let's show the business impact!\ndefault_preds = (y_test_probs >= 0.5).astype(int)\nfrom sklearn.metrics import recall_score, precision_score\nprint(\"\\n--- BUSINESS IMPACT SUMMARY ---\")\nprint(f\"Default (0.5) Precision:  {precision_score(y_test, default_preds):.3f}\")\nprint(f\"Default (0.5) Recall:     {recall_score(y_test, default_preds):.3f}\")\nprint(\"-\" * 30)\nprint(f\"Optimized Precision:      {precision_score(y_test, y_pred_optimized):.3f}\")\nprint(f\"Optimized Recall:         {recall_score(y_test, y_pred_optimized):.3f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T10:06:15.135667Z","iopub.execute_input":"2026-05-07T10:06:15.136040Z","iopub.status.idle":"2026-05-07T10:06:15.186716Z","shell.execute_reply.started":"2026-05-07T10:06:15.136006Z","shell.execute_reply":"2026-05-07T10:06:15.185011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 7: THE ACTION & RECOMMENDATION ENGINE ---\nimport pandas as pd\nimport random\n\nclass RetentionActionEngine:\n    def __init__(self, ai_model, threshold):\n        self.model = ai_model\n        self.threshold = threshold\n        \n        # Define what constitutes a \"High Value\" customer in this dataset\n        # (e.g., if they pay more than 500 NTD on average)\n        self.VIP_VALUE_THRESHOLD = 500.0 \n\n    def analyze_user(self, user_id, user_features):\n        \"\"\"Processes a single user and returns the complete retention strategy.\"\"\"\n        \n        # 1. Predict Risk\n        # We must drop user_id if it's in the features, just like training\n        features_for_ai = user_features.drop(labels=['msno'], errors='ignore')\n        risk_prob = self.model.predict_proba([features_for_ai])[0][1]\n        \n        # Extract basic info for personalization\n        avg_spend = user_features['avg_plan_price']\n        tenure = user_features['billing_tenure_days']\n        \n        # 2. Value Assessment\n        is_vip = avg_spend >= self.VIP_VALUE_THRESHOLD\n        \n        # 3. Decision Logic\n        if risk_prob < self.threshold:\n            decision = \"NO_ACTION\"\n            urgency = \"Low\"\n            recommendation = \"Customer is safe. Continue normal service.\"\n            \n        elif risk_prob >= self.threshold and not is_vip:\n            decision = \"AUTOMATED_EMAIL\"\n            urgency = \"Medium\"\n            recommendation = self._generate_email_recommendation(tenure)\n            \n        elif risk_prob >= self.threshold and is_vip:\n            decision = \"HUMAN_ESCALATION\"\n            urgency = \"CRITICAL\"\n            recommendation = self._generate_sales_script(tenure, avg_spend)\n            \n        return {\n            \"User_ID\": user_id,\n            \"Churn_Risk\": f\"{risk_prob * 100:.1f}%\",\n            \"Urgency\": urgency,\n            \"Action_Type\": decision,\n            \"Recommendation_Script\": recommendation\n        }\n\n    def _generate_email_recommendation(self, tenure):\n        \"\"\"Dynamic text generator for automated emails.\"\"\"\n        if tenure > 365:\n            return \"Email Template A: 'Happy Anniversary! Thanks for being with us for over a year. Here is 20% off your next month to celebrate!'\"\n        else:\n            return \"Email Template B: 'We miss you! Come back and explore our new premium features with a 10% discount.'\"\n\n    def _generate_sales_script(self, tenure, spend):\n        \"\"\"Dynamic text generator for human sales agents.\"\"\"\n        return f\"URGENT TASK FOR ACCOUNT MGR: Call this VIP user immediately. They spend {spend} and have been with us for {tenure} days. Offer a free 1-on-1 strategy call and a complimentary upgrade to keep them.\"\n\n\nprint(\"1. Initializing the Action Engine with our AI Model and Optimal Threshold...\")\n# Note: Ensure you use the final_model from your Auto-Tuning cell!\naction_engine = RetentionActionEngine(ai_model=final_model, threshold=0.433)\n\nprint(\"2. Simulating Production: Running 5 Random Test Users through the Engine...\")\n# We will grab 5 random users from our test set\nsample_users = X_test.sample(5, random_state=42)\n\n# We need to map the 'msno' IDs back to these users for the report\nsample_user_ids = final_matrix.loc[sample_users.index, 'msno'].values\n\nresults = []\nfor i in range(len(sample_users)):\n    user_features = sample_users.iloc[i]\n    user_id = sample_user_ids[i]\n    \n    # Run the engine\n    strategy = action_engine.analyze_user(user_id, user_features)\n    results.append(strategy)\n\n# Display the final business report\nreport_df = pd.DataFrame(results)\ndisplay(report_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T10:06:19.918148Z","iopub.execute_input":"2026-05-07T10:06:19.918568Z","iopub.status.idle":"2026-05-07T10:06:19.950165Z","shell.execute_reply.started":"2026-05-07T10:06:19.918470Z","shell.execute_reply":"2026-05-07T10:06:19.949008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CELL 8: DEEP ANALYSIS & BILINGUAL LLAMA AI EXPLANATION ---\nimport shap\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport requests\nimport json\n\n# 1. Initialize SHAP Javascript visualization in Kaggle\nshap.initjs() \n\n# 🛑 Insert your LLaMA API Provider details here 🛑\n# Using Groq as the default provider for blazing fast LLaMA 3\nLLAMA_API_KEY = \"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\"\nLLAMA_API_URL = \"https://api.groq.com/openai/v1/chat/completions\"\n# Change this line:\nLLAMA_MODEL_NAME = \"llama-3.3-70b-versatile\" # Or 'meta-llama/Llama-3-70b-chat-hf' if using Together AI\n\nprint(\"1. Initializing SHAP Tree Explainer...\")\nexplainer = shap.TreeExplainer(final_model)\n\n# Feature translation dictionary (Used to feed the LLM readable context)\nFEATURE_TRANSLATIONS = {\n    'total_transactions': 'Total Transactions (Frequency)',\n    'total_cancellations': 'Previous Cancellations',\n    'auto_renew_count': 'Auto-Renew Usage',\n    'total_amount_paid': 'Total Amount Paid (Monetary)',\n    'avg_plan_price': 'Average Plan Price',\n    'billing_tenure_days': 'Billing Tenure (Days)',\n    'cancel_rate': 'Cancellation Rate'\n}\n\nclass LlamaRetentionAnalyzer:\n    def __init__(self, ai_model, shap_explainer, api_key, api_url, model_name):\n        self.model = ai_model\n        self.explainer = shap_explainer\n        self.api_key = api_key\n        self.api_url = api_url\n        self.model_name = model_name\n\n    def analyze_and_explain(self, user_id, user_features):\n        \"\"\"Generates SHAP analysis and uses LLaMA to write a bilingual explanation.\"\"\"\n        \n        print(f\"\\n{'='*60}\")\n        print(f\"🤖 AI AGENT ANALYSIS FOR USER: {user_id}\")\n        print(f\"{'='*60}\")\n\n        # 1. Predict Churn Risk\n        features_df = user_features.drop(labels=['msno', 'is_churn'], errors='ignore').to_frame().T\n        risk_prob = self.model.predict_proba(features_df)[0][1]\n        risk_percentage = risk_prob * 100\n        print(f\"⚠️ Predicted Churn Risk: {risk_percentage:.1f}%\")\n        \n        # 2. Calculate exact SHAP values for this specific user\n        shap_values = self.explainer.shap_values(features_df)\n        \n        # 3. Extract the top negative drivers (Features pushing the user to churn)\n        drivers = []\n        for i, col in enumerate(features_df.columns):\n            impact = shap_values[0][i]\n            if impact > 0: # We only care about positive impact values (which increase churn risk)\n                drivers.append({\n                    \"feature\": FEATURE_TRANSLATIONS.get(col, col),\n                    \"value\": features_df.iloc[0, i],\n                    \"impact\": impact\n                })\n                \n        # Sort drivers from highest impact to lowest\n        drivers = sorted(drivers, key=lambda x: x['impact'], reverse=True)\n        \n        # Format the top 3 drivers into a text string for the LLM\n        top_drivers_text = \"\\n\".join([f\"- {d['feature']} (Current Value: {d['value']})\" for d in drivers[:3]])\n\n        # 4. Call LLaMA API to translate numbers into a bilingual human narrative\n        if self.api_key != \"gsk-your-api-key-here\":\n            print(\"⏳ Calling LLaMA API to generate the bilingual narrative report...\\n\")\n            llama_report = self._call_llama_api(user_id, risk_percentage, top_drivers_text)\n            \n            print(\"✨ [LLaMA Smart Report for Customer Success] ✨\\n\")\n            print(llama_report)\n        else:\n            print(\"⚠️ Warning: Please insert your LLaMA API key to see the narrative report.\")\n            print(\"\\nExtracted Root Causes (SHAP Drivers):\")\n            print(top_drivers_text)\n\n        # 5. Display the visual Force Plot\n        print(\"\\n📈 [SHAP Force Plot - Visual Analysis]\")\n        plot = shap.force_plot(\n            base_value=self.explainer.expected_value, \n            shap_values=shap_values[0,:], \n            features=features_df.iloc[0,:],\n            feature_names=[FEATURE_TRANSLATIONS.get(c, c) for c in features_df.columns],\n            matplotlib=True,\n            show=False\n        )\n        plt.gcf().set_size_inches(16, 4)\n        plt.tight_layout()\n        plt.show()\n\n    def _call_llama_api(self, user_id, risk_percentage, top_drivers_text):\n        \"\"\"Sends SHAP insights to LLaMA with advanced Prompt Engineering for maximum readability.\"\"\"\n        \n        # --- THE ENHANCED PROMPT ---\n        prompt = f\"\"\"\nYou are an elite AI Customer Retention Strategist working for a billion-dollar subscription company.\n\nWe have detected a HIGH-RISK VIP customer.\n\n━━━━━━━━━━━━━━━━━━━━━━━\n📌 CUSTOMER PROFILE\n━━━━━━━━━━━━━━━━━━━━━━━\n• Customer ID: {user_id}\n• Predicted Churn Risk: {risk_percentage:.1f}%\n• Risk Level: {\"CRITICAL\" if risk_percentage >= 85 else \"HIGH\"}\n\n━━━━━━━━━━━━━━━━━━━━━━━\n📊 SHAP ROOT-CAUSE ANALYSIS\n━━━━━━━━━━━━━━━━━━━━━━━\nThe ML model identified these behaviors as the strongest churn drivers:\n{top_drivers_text}\n\n━━━━━━━━━━━━━━━━━━━━━━━\n🎯 YOUR OBJECTIVE\n━━━━━━━━━━━━━━━━━━━━━━━\nGenerate a concise, highly professional, psychologically intelligent retention report for a HUMAN customer-success agent.\n\nThe report must NOT sound robotic.\n\nThe report must explain:\n• WHAT is happening\n• WHY the customer is behaving this way\n• WHAT emotional/business signals are visible\n• WHAT intervention has the highest probability of saving the customer\n\n━━━━━━━━━━━━━━━━━━━━━━━\n🧠 DEEP ANALYSIS REQUIREMENTS\n━━━━━━━━━━━━━━━━━━━━━━━\n\nYou MUST:\n1. Interpret the behavioral meaning behind the SHAP features.\n2. Infer possible customer frustration, disengagement, dissatisfaction, or hesitation.\n3. Explain WHY these behaviors increase churn probability.\n4. Connect the behavioral signals together into one coherent story.\n5. Prioritize the MOST important risk factors only.\n6. Avoid generic AI wording.\n7. Keep explanations simple enough for non-technical business agents.\n8. Sound like a senior retention strategist.\n\n━━━━━━━━━━━━━━━━━━━━━━━\n🎨 STRICT OUTPUT STYLE\n━━━━━━━━━━━━━━━━━━━━━━━\n\n• Use BEAUTIFUL Markdown formatting\n• Use section headers\n• Use bullet points\n• Use bold text heavily\n• Use emojis intelligently (🚨 🔍 💡 📉 📞 ❤️)\n• Keep paragraphs SHORT\n• Be concise but insightful\n• NO fluff\n• NO repeated information\n• Make the report highly scannable\n\n━━━━━━━━━━━━━━━━━━━━━━━\n🌐 BILINGUAL REQUIREMENT\n━━━━━━━━━━━━━━━━━━━━━━━\n\nYou MUST:\n1. Write the FULL report in ARABIC first\n2. Insert this separator exactly:\n--------------------------------------------------\n3. Then provide the COMPLETE ENGLISH version\n\nThe Arabic must sound natural and professional, not machine translated.\n\n━━━━━━━━━━━━━━━━━━━━━━━\n📑 REQUIRED REPORT STRUCTURE\n━━━━━━━━━━━━━━━━━━━━━━━\n\n# 🚨 Churn Risk Summary\n- Briefly summarize the danger level.\n- Mention the predicted risk percentage.\n- Explain how urgent the situation is.\n\n# 🔍 Behavioral Diagnosis\n- Explain the behavioral patterns detected.\n- Explain WHY these patterns are dangerous.\n- Explain what the customer is likely experiencing psychologically or behaviorally.\n\n# 📉 Root Causes Ranked\nRank the top churn drivers from most dangerous to least dangerous.\nFor EACH driver:\n- Explain what it means\n- Why it matters\n- How it contributes to churn\n\n# 💡 Recommended Rescue Strategy\nProvide:\n- Best retention offer\n- Recommended communication tone\n- Suggested incentive\n- Priority level\n\n# ❤️ Empathy Guidance\nExplain how the human agent should emotionally approach this customer.\n\n# 📞 Suggested Agent Script\nWrite ONE short, natural, empathetic opening sentence the agent can say immediately on a phone call.\n\n# ⚡ Executive Takeaway\nProvide a final 1-sentence executive summary.\n\"\"\"\n\n        headers = {\n            \"Content-Type\": \"application/json\",\n            \"Authorization\": f\"Bearer {self.api_key}\"\n        }\n        \n        data = {\n            \"model\": self.model_name, \n            \"messages\": [\n                {\n                    \"role\": \"system\", \n                    \"content\": \"You are a concise, brilliant Customer Success strategist. You format all responses beautifully with Markdown.\"\n                },\n                {\"role\": \"user\", \"content\": prompt}\n            ],\n            # Lowering temperature slightly to 0.5 makes the AI follow formatting rules more strictly\n            \"temperature\": 0.5 \n        }\n\n        try:\n            response = requests.post(self.api_url, headers=headers, json=data)\n            response.raise_for_status()\n            result = response.json()\n            return result['choices'][0]['message']['content']\n        except requests.exceptions.HTTPError as e:\n            return f\"API HTTP Error: {e}\\nDetails: {e.response.text}\"\n        except Exception as e:\n            return f\"General Error connecting to LLaMA API: {e}\"\n\n# --- Execution ---\nprint(\"\\n2. Initializing LLaMA-Powered Deep Analyzer...\")\nllama_analyzer = LlamaRetentionAnalyzer(\n    ai_model=final_model, \n    shap_explainer=explainer, \n    api_key=LLAMA_API_KEY,\n    api_url=LLAMA_API_URL,\n    model_name=LLAMA_MODEL_NAME\n)\n\nprint(\"\\n3. Hunting for a High-Risk VIP User to analyze...\\n\")\nfor i in range(len(X_test)):\n    features = X_test.iloc[i]\n    features_for_pred = features.drop(labels=['msno', 'is_churn'], errors='ignore').to_frame().T\n    \n    # Trigger the analysis if the user has an 80%+ risk of churning\n    if final_model.predict_proba(features_for_pred)[0][1] > 0.80:\n        user_id = final_matrix.loc[X_test.index[i], 'msno']\n        llama_analyzer.analyze_and_explain(user_id, features)\n        break # Process only one user for this demonstration","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-07T10:06:31.358120Z","iopub.execute_input":"2026-05-07T10:06:31.358623Z","iopub.status.idle":"2026-05-07T10:06:35.510069Z","shell.execute_reply.started":"2026-05-07T10:06:31.358586Z","shell.execute_reply":"2026-05-07T10:06:35.508996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}