{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":104206,"databundleVersionId":12563446,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:26.830153Z","iopub.execute_input":"2025-07-10T15:50:26.830708Z","iopub.status.idle":"2025-07-10T15:50:27.214209Z","shell.execute_reply.started":"2025-07-10T15:50:26.830680Z","shell.execute_reply":"2025-07-10T15:50:27.213319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\ngc.collect()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:27.215739Z","iopub.execute_input":"2025-07-10T15:50:27.216171Z","iopub.status.idle":"2025-07-10T15:50:27.284719Z","shell.execute_reply.started":"2025-07-10T15:50:27.216148Z","shell.execute_reply":"2025-07-10T15:50:27.283909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_DIR = \"/kaggle/input/social-sim-challenge-social-media-based-personas/train\"\nOUTPUT_PATH = \"/kaggle/working/train_dataset.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:27.285400Z","iopub.execute_input":"2025-07-10T15:50:27.285728Z","iopub.status.idle":"2025-07-10T15:50:27.301935Z","shell.execute_reply.started":"2025-07-10T15:50:27.285707Z","shell.execute_reply":"2025-07-10T15:50:27.300951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nos.listdir(DATA_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:27.302721Z","iopub.execute_input":"2025-07-10T15:50:27.302951Z","iopub.status.idle":"2025-07-10T15:50:27.322000Z","shell.execute_reply.started":"2025-07-10T15:50:27.302917Z","shell.execute_reply":"2025-07-10T15:50:27.321217Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport ujson\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom typing import List, Dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:27.324171Z","iopub.execute_input":"2025-07-10T15:50:27.324436Z","iopub.status.idle":"2025-07-10T15:50:27.341219Z","shell.execute_reply.started":"2025-07-10T15:50:27.324395Z","shell.execute_reply":"2025-07-10T15:50:27.340470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os, ujson, pandas as pd\nfrom tqdm import tqdm\nimport gc\n\nDATA_DIR = \"/kaggle/input/social-sim-challenge-social-media-based-personas/train\"\nSAVE_DIR = \"./parsed_clusters\"\nos.makedirs(SAVE_DIR, exist_ok=True)\n\nlabel_cols = ['like', 'unlike', 'repost', 'unrepost', 'follow', 'unfollow', 'block',\n              'unblock', 'post_update', 'post_delete', 'quote', 'post', 'reply']\n\ndef extract_from_cluster(filepath):\n    rows = []\n    with open(filepath, \"r\") as f:\n        for line in f:\n            obj = ujson.loads(line)\n            cluster_id = obj[\"cluster_id\"]\n            thread_text = []\n            row = {\"cluster_id\": cluster_id, \"id\": obj[\"id\"]}\n            for entry in obj[\"thread\"]:\n                if \"text\" in entry:\n                    thread_text.append(entry[\"text\"])\n                if \"actions\" in entry:\n                    for action in label_cols:\n                        row[f\"label_{action}\"] = entry[\"actions\"].get(action, False)\n            row[\"history_text\"] = \"\\n\".join(thread_text)\n            rows.append(row)\n    return rows\n\n# ✅ Stream and save per-cluster safely\nfor i in tqdm(range(25)):\n    file_path = os.path.join(DATA_DIR, f\"cluster_{i}.jsonl\")\n    cluster_data = extract_from_cluster(file_path)\n    df = pd.DataFrame(cluster_data)\n    df.to_csv(f\"{SAVE_DIR}/cluster_{i}_parsed.csv\", index=False)\n    del df, cluster_data\n    gc.collect()\n\nprint(\"✅ Parsed and saved all clusters to ./parsed_clusters/\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:50:27.342173Z","iopub.execute_input":"2025-07-10T15:50:27.342500Z","iopub.status.idle":"2025-07-10T15:54:20.705422Z","shell.execute_reply.started":"2025-07-10T15:50:27.342473Z","shell.execute_reply":"2025-07-10T15:54:20.704478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\ndfs = [pd.read_csv(file) for file in glob.glob(\"./parsed_clusters/*.csv\")]\ndf_all = pd.concat(dfs, ignore_index=True)\ndf_all.to_csv(\"final_labels_with_history.csv\", index=False)\nprint(\"✅ final_labels_with_history.csv saved. Shape:\", df_all.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:54:20.706619Z","iopub.execute_input":"2025-07-10T15:54:20.706861Z","iopub.status.idle":"2025-07-10T15:56:43.183225Z","shell.execute_reply.started":"2025-07-10T15:54:20.706842Z","shell.execute_reply":"2025-07-10T15:56:43.182396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport pandas as pd\n\ncsv_files = glob.glob(\"./parsed_clusters/*.csv\")\ndfs = [pd.read_csv(f) for f in csv_files]\ndf_all = pd.concat(dfs, ignore_index=True)\nprint(df_all.shape)\nprint(df_all.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:56:43.184170Z","iopub.execute_input":"2025-07-10T15:56:43.184445Z","iopub.status.idle":"2025-07-10T15:57:31.187645Z","shell.execute_reply.started":"2025-07-10T15:56:43.184395Z","shell.execute_reply":"2025-07-10T15:57:31.186742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# 1. Distribution of all 13 actions\nlabel_cols = [col for col in df_all.columns if col.startswith(\"label_\")]\naction_counts = df_all[label_cols].sum().sort_values(ascending=False)\n\nplt.figure(figsize=(12,6))\nsns.barplot(x=action_counts.values, y=action_counts.index, palette=\"Blues_r\")\nplt.title(\"Action Frequency Distribution\")\nplt.xlabel(\"Count\")\nplt.ylabel(\"Action\")\nplt.grid(axis=\"x\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:57:31.188753Z","iopub.execute_input":"2025-07-10T15:57:31.189014Z","iopub.status.idle":"2025-07-10T15:57:33.047626Z","shell.execute_reply.started":"2025-07-10T15:57:31.188993Z","shell.execute_reply":"2025-07-10T15:57:33.046739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import multilabel_confusion_matrix\nimport numpy as np\n\n# Compute correlation matrix\nco_matrix = df_all[label_cols].T.dot(df_all[label_cols])\nnp.fill_diagonal(co_matrix.values, 0)  # remove self-cooccurrence\n\nplt.figure(figsize=(10, 8))\nsns.heatmap(co_matrix, annot=True, fmt=\"d\", cmap=\"coolwarm\", square=True)\nplt.title(\"Action Co-occurrence Heatmap\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:57:33.048560Z","iopub.execute_input":"2025-07-10T15:57:33.049014Z","iopub.status.idle":"2025-07-10T15:57:37.248128Z","shell.execute_reply.started":"2025-07-10T15:57:33.048990Z","shell.execute_reply":"2025-07-10T15:57:37.247296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all[\"history_len\"] = df_all[\"history_text\"].str.split().apply(len)\n\nplt.figure(figsize=(10, 4))\nsns.histplot(df_all[\"history_len\"], bins=100, kde=True, color=\"teal\")\nplt.title(\"Distribution of History Text Length (in words)\")\nplt.xlabel(\"Number of words\")\nplt.ylabel(\"Count\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:57:37.249262Z","iopub.execute_input":"2025-07-10T15:57:37.249686Z","iopub.status.idle":"2025-07-10T15:59:31.502814Z","shell.execute_reply.started":"2025-07-10T15:57:37.249656Z","shell.execute_reply":"2025-07-10T15:59:31.501908Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all[\"label_sum\"] = df_all[label_cols].sum(axis=1)\n\nplt.figure(figsize=(8,4))\nsns.countplot(x=\"label_sum\", data=df_all, palette=\"viridis\")\nplt.title(\"How Many Actions Co-occur?\")\nplt.xlabel(\"Number of Actions (per row)\")\nplt.ylabel(\"Count\")\nplt.show()\n\ndf_all[\"label_sum\"].value_counts(normalize=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:31.503814Z","iopub.execute_input":"2025-07-10T15:59:31.504062Z","iopub.status.idle":"2025-07-10T15:59:33.379111Z","shell.execute_reply.started":"2025-07-10T15:59:31.504041Z","shell.execute_reply":"2025-07-10T15:59:33.378277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"action_avg_lengths = {\n    action: df_all[df_all[action]][\"history_len\"].mean()\n    for action in label_cols\n}\npd.Series(action_avg_lengths).sort_values(ascending=False).plot(kind=\"barh\", figsize=(10,6), color='orange')\nplt.title(\"Average History Length by Action Type\")\nplt.xlabel(\"Avg # of Words in History\")\nplt.ylabel(\"Action Type\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:33.380176Z","iopub.execute_input":"2025-07-10T15:59:33.381050Z","iopub.status.idle":"2025-07-10T15:59:34.968568Z","shell.execute_reply.started":"2025-07-10T15:59:33.381026Z","shell.execute_reply":"2025-07-10T15:59:34.967693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count labels by cluster\ncluster_action_counts = df_all.groupby(\"cluster_id\")[label_cols].sum()\n\nplt.figure(figsize=(14, 8))\nsns.heatmap(cluster_action_counts, annot=False, cmap=\"YlGnBu\", linewidths=0.3)\nplt.title(\"Cluster-wise Action Frequency\")\nplt.xlabel(\"Action\")\nplt.ylabel(\"Persona Cluster\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:34.972660Z","iopub.execute_input":"2025-07-10T15:59:34.972953Z","iopub.status.idle":"2025-07-10T15:59:36.029584Z","shell.execute_reply.started":"2025-07-10T15:59:34.972931Z","shell.execute_reply":"2025-07-10T15:59:36.028715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nimport time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:36.030517Z","iopub.execute_input":"2025-07-10T15:59:36.030803Z","iopub.status.idle":"2025-07-10T15:59:36.174098Z","shell.execute_reply.started":"2025-07-10T15:59:36.030773Z","shell.execute_reply":"2025-07-10T15:59:36.173354Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df_all[\"history_text\"]\ny = df_all[label_cols]\n\nfrom sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:36.174962Z","iopub.execute_input":"2025-07-10T15:59:36.175612Z","iopub.status.idle":"2025-07-10T15:59:40.261145Z","shell.execute_reply.started":"2025-07-10T15:59:36.175582Z","shell.execute_reply":"2025-07-10T15:59:40.260318Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# STEP 1: Keep only samples with at least one positive label\nmask = y_train[label_cols].sum(axis=1) > 0\nX_pos = X_train[mask]\ny_pos = y_train[mask]\n\n# STEP 2: Take first 20000 rows\nX_train_sub = X_pos[:20000]\ny_train_sub = y_pos.iloc[:20000]\n\n# STEP 3: Drop label columns that still have only one class (e.g., all 0s)\nlabel_cols_sub = [col for col in label_cols if len(y_train_sub[col].unique()) > 1]\ny_train_sub = y_train_sub[label_cols_sub]\n\n# STEP 4: Vectorize text\ntfidf = TfidfVectorizer(max_features=20000, stop_words=\"english\")\nX_train_tfidf_sub = tfidf.fit_transform(X_train_sub)\nX_val_tfidf = tfidf.transform(X_val)\n\n# STEP 5: Fit model\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nimport time\n\nbase_model = LogisticRegression(class_weight=\"balanced\", solver='liblinear', max_iter=1000)\nmulti_model = MultiOutputClassifier(base_model)\n\nstart = time.time()\nmulti_model.fit(X_train_tfidf_sub, y_train_sub)\nprint(f\"✅ Subset training completed in {(time.time() - start)/60:.2f} minutes\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T15:59:40.262097Z","iopub.execute_input":"2025-07-10T15:59:40.262343Z","iopub.status.idle":"2025-07-10T16:00:21.565304Z","shell.execute_reply.started":"2025-07-10T15:59:40.262324Z","shell.execute_reply":"2025-07-10T16:00:21.563585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\n# Predict on validation set\ny_pred = multi_model.predict(X_val_tfidf)\n\n# Restrict y_val to the same label columns used in training\ny_val_filtered = y_val[label_cols_sub]\n\n# Compute F1 scores\nprint(\"📊 Per-label F1 Scores:\")\nfor i, col in enumerate(label_cols_sub):\n    f1 = f1_score(y_val_filtered[col], y_pred[:, i], average='binary', zero_division=0)\n    print(f\"{col:20s}: F1 = {f1:.4f}\")\n\n# Overall macro F1\noverall_macro_f1 = f1_score(y_val_filtered, y_pred, average='macro', zero_division=0)\nprint(f\"\\n🔥 Overall Macro F1 Score: {overall_macro_f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:00:21.566056Z","iopub.execute_input":"2025-07-10T16:00:21.566311Z","iopub.status.idle":"2025-07-10T16:00:29.953755Z","shell.execute_reply.started":"2025-07-10T16:00:21.566288Z","shell.execute_reply":"2025-07-10T16:00:29.949226Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3","metadata":{}},{"cell_type":"code","source":"!pip install -q sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\nimport time","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:00:29.954877Z","iopub.execute_input":"2025-07-10T16:00:29.955279Z","iopub.status.idle":"2025-07-10T16:02:33.847875Z","shell.execute_reply.started":"2025-07-10T16:00:29.955246Z","shell.execute_reply":"2025-07-10T16:02:33.846836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_all = pd.read_csv(\"final_labels_with_history.csv\")  # or whatever file you saved\nlabel_cols = [col for col in df_all.columns if col.startswith(\"label_\")]\n\nX = df_all[\"history_text\"]\ny = df_all[label_cols]\n\n# Train/Val Split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:32:00.272884Z","iopub.execute_input":"2025-07-10T16:32:00.273253Z","iopub.status.idle":"2025-07-10T16:32:55.077224Z","shell.execute_reply.started":"2025-07-10T16:32:00.273218Z","shell.execute_reply":"2025-07-10T16:32:55.076105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sentence_transformers import SentenceTransformer\n\nmodel = SentenceTransformer('all-MiniLM-L6-v2')  # Initialize model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:32:55.093815Z","iopub.execute_input":"2025-07-10T16:32:55.094145Z","iopub.status.idle":"2025-07-10T16:32:55.796605Z","shell.execute_reply.started":"2025-07-10T16:32:55.094121Z","shell.execute_reply":"2025-07-10T16:32:55.795801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Filter samples with sufficient history length\nmin_length = 300\nmask = df_all['history_text'].str.len() > min_length\n\nX = df_all.loc[mask, 'history_text']\ny = df_all.loc[mask, label_cols]\n\n# Split train/val\n# (Assuming you have a split, else do train_test_split here)\n\n# Subset training data to 20k samples for faster prototyping\nX_train_sub = X_train[:20000]\ny_train_sub = y_train.iloc[:20000]\n\n# Filter labels with >=2 classes in train subset\nlabel_cols_sub = [col for col in y_train_sub.columns if len(y_train_sub[col].unique()) > 1]\ny_train_sub_filtered = y_train_sub[label_cols_sub]\n\n# Use filtered labels for training and evaluation\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:32:55.816979Z","iopub.execute_input":"2025-07-10T16:32:55.817607Z","iopub.status.idle":"2025-07-10T16:33:01.371483Z","shell.execute_reply.started":"2025-07-10T16:32:55.817576Z","shell.execute_reply":"2025-07-10T16:33:01.370764Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MAX_WORDS = 300\ndf_all[\"history_text\"] = df_all[\"history_text\"].apply(lambda x: \" \".join(x.split()[:MAX_WORDS]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:33:01.381760Z","iopub.execute_input":"2025-07-10T16:33:01.382195Z","iopub.status.idle":"2025-07-10T16:33:29.177222Z","shell.execute_reply.started":"2025-07-10T16:33:01.382164Z","shell.execute_reply":"2025-07-10T16:33:29.176363Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install -q sentence-transformers\n\nfrom sentence_transformers import SentenceTransformer\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\nimport time\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:33:29.187544Z","iopub.execute_input":"2025-07-10T16:33:29.187915Z","iopub.status.idle":"2025-07-10T16:33:33.634465Z","shell.execute_reply.started":"2025-07-10T16:33:29.187883Z","shell.execute_reply":"2025-07-10T16:33:33.633260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.metrics import f1_score\nfrom sentence_transformers import SentenceTransformer\n\n# ✂️ Step 1: Filter samples with long enough history (length > 300)\ndf_filtered = df_all[df_all[\"history_text\"].str.len() > 300].reset_index(drop=True)\n\n# 🧾 Step 2: Split X and y\nX = df_filtered[\"history_text\"]\ny = df_filtered[[col for col in df_filtered.columns if col.startswith(\"label_\")]]\n\n# ✂️ Step 3: Subset first 20k samples for train and 4k for val\nX_train = X[:20000]\ny_train = y.iloc[:20000]\nX_val = X[20000:24000]\ny_val = y.iloc[20000:24000]\n\n# 🧹 Step 4: Remove labels that are all one class (e.g., all False) in train\nlabel_cols_sub = [col for col in y_train.columns if len(y_train[col].unique()) > 1]\ny_train = y_train[label_cols_sub]\ny_val = y_val[label_cols_sub]\n\n# 💠 Step 5: Embed text\nmodel = SentenceTransformer('all-MiniLM-L6-v2')\nX_train_embeds = model.encode(X_train.tolist(), batch_size=64, show_progress_bar=True)\nX_val_embeds = model.encode(X_val.tolist(), batch_size=64, show_progress_bar=True)\n\n# 🤖 Step 6: Train and predict\nbase_model = LogisticRegression(class_weight='balanced', solver='liblinear', max_iter=1000)\nmulti_model = MultiOutputClassifier(base_model)\n\nmulti_model.fit(X_train_embeds, y_train)\ny_pred = multi_model.predict(X_val_embeds)\n\nprint(\"📊 Per-label F1 Scores:\")\nfor i, col in enumerate(label_cols_sub):  # Use only trained labels\n    f1 = f1_score(y_val[col], y_pred[:, i], average='binary', zero_division=0)\n    print(f\"{col:20s}: F1 = {f1:.4f}\")\n\n\nmacro_f1 = f1_score(y_val, y_pred, average='macro', zero_division=0)\nprint(f\"\\n🔥 Macro F1 Score (subset): {macro_f1:.4f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:33:33.646004Z","iopub.execute_input":"2025-07-10T16:33:33.646304Z","iopub.status.idle":"2025-07-10T16:47:38.588591Z","shell.execute_reply.started":"2025-07-10T16:33:33.646281Z","shell.execute_reply":"2025-07-10T16:47:38.587497Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"3.2","metadata":{}},{"cell_type":"code","source":"!pip install -q transformers datasets\n\nimport pandas as pd\nimport torch\nfrom transformers import GPT2Tokenizer, GPT2LMHeadModel, Trainer, TrainingArguments, DataCollatorForLanguageModeling\nfrom datasets import Dataset\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:38.619359Z","iopub.execute_input":"2025-07-10T16:47:38.619669Z","iopub.status.idle":"2025-07-10T16:47:43.221222Z","shell.execute_reply.started":"2025-07-10T16:47:38.619648Z","shell.execute_reply":"2025-07-10T16:47:43.220063Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_filtered is already loaded if you're continuing; if not, reload it:\n# df_filtered = pd.read_csv(\"final_labels_with_history.csv\")\n# df_filtered = df_filtered[df_filtered[\"history_text\"].str.len() > 300].reset_index(drop=True)\n\ndf_gen = df_filtered[df_filtered[\"label_post\"] == True][[\"history_text\"]].copy()\ndf_gen[\"prompt\"] = \"Given the following social history, generate a response:\\n\" + df_gen[\"history_text\"]\n\ndef format_prompt(example):\n    return {\n        \"input_text\": f\"[Persona: {example['persona']}] [Context: {example['context']}] =>\",\n        \"label_text\": example['post_text']\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:43.236068Z","iopub.execute_input":"2025-07-10T16:47:43.236561Z","iopub.status.idle":"2025-07-10T16:47:43.278443Z","shell.execute_reply.started":"2025-07-10T16:47:43.236528Z","shell.execute_reply":"2025-07-10T16:47:43.277530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import DataCollatorForLanguageModeling\ntokenizer = GPT2Tokenizer.from_pretrained(\"distilgpt2\")\ntokenizer.pad_token = tokenizer.eos_token\ndata_collator = DataCollatorForLanguageModeling(\n    tokenizer=tokenizer, \n    mlm=False  # Important: we are NOT doing masked language modeling (like BERT)\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:43.290538Z","iopub.execute_input":"2025-07-10T16:47:43.290989Z","iopub.status.idle":"2025-07-10T16:47:43.633628Z","shell.execute_reply.started":"2025-07-10T16:47:43.290964Z","shell.execute_reply":"2025-07-10T16:47:43.632696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = GPT2Tokenizer.from_pretrained(\"distilgpt2\")\ntokenizer.pad_token = tokenizer.eos_token\n\ndef tokenize_function(examples):\n    return tokenizer(\n        examples[\"prompt\"],\n        truncation=True,\n        max_length=512,\n        padding=\"max_length\"\n    )\n\ndataset = Dataset.from_pandas(df_gen[[\"prompt\"]])\ntokenized_dataset = dataset.map(tokenize_function, batched=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:43.645488Z","iopub.execute_input":"2025-07-10T16:47:43.645869Z","iopub.status.idle":"2025-07-10T16:47:47.612828Z","shell.execute_reply.started":"2025-07-10T16:47:43.645847Z","shell.execute_reply":"2025-07-10T16:47:47.611808Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import AutoModelForCausalLM\n\nmodel = AutoModelForCausalLM.from_pretrained(\"distilgpt2\")\nmodel.resize_token_embeddings(len(tokenizer))\n\ndata_collator = DataCollatorForLanguageModeling(\n    tokenizer=tokenizer, mlm=False\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:47.630681Z","iopub.execute_input":"2025-07-10T16:47:47.630909Z","iopub.status.idle":"2025-07-10T16:47:47.837358Z","shell.execute_reply.started":"2025-07-10T16:47:47.630891Z","shell.execute_reply":"2025-07-10T16:47:47.836432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\n\ndef clean_text(text):\n    text = re.sub(r\"<.*?>\", \"\", text)  # remove HTML\n    text = re.sub(r\"http\\S+\", \"\", text)  # remove URLs\n    return text.strip()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:47.859517Z","iopub.execute_input":"2025-07-10T16:47:47.859788Z","iopub.status.idle":"2025-07-10T16:47:47.864450Z","shell.execute_reply.started":"2025-07-10T16:47:47.859761Z","shell.execute_reply":"2025-07-10T16:47:47.863544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(type(tokenized_dataset))  # Should print: <class 'datasets.arrow_dataset.Dataset'>\nprint(tokenized_dataset.column_names)  # Should include 'input_ids', 'attention_mask', etc.\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:47.865612Z","iopub.execute_input":"2025-07-10T16:47:47.865867Z","iopub.status.idle":"2025-07-10T16:47:47.882777Z","shell.execute_reply.started":"2025-07-10T16:47:47.865848Z","shell.execute_reply":"2025-07-10T16:47:47.881817Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from transformers import Trainer, TrainingArguments\n\ntraining_args = TrainingArguments(\n    output_dir=\"./distilgpt2-socialsim\",     \n    overwrite_output_dir=True,               \n    per_device_train_batch_size=4,          \n    num_train_epochs=3,                     \n    max_steps=1,                          \n    logging_steps=10,                       \n    save_steps=50,                         \n    save_total_limit=2,                      \n    fp16=True,                               \n    disable_tqdm=False,                      \n    report_to=[],                        \n    run_name=\"test-socialsim-gpt2\"           \n)\n\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_dataset,  # ✅ fix here\n    tokenizer=tokenizer,\n    data_collator=data_collator,\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:47.883858Z","iopub.execute_input":"2025-07-10T16:47:47.884143Z","iopub.status.idle":"2025-07-10T16:47:48.156615Z","shell.execute_reply.started":"2025-07-10T16:47:47.884112Z","shell.execute_reply":"2025-07-10T16:47:48.155738Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"trainer.train()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:47:48.172873Z","iopub.execute_input":"2025-07-10T16:47:48.173256Z","iopub.status.idle":"2025-07-10T16:48:06.770822Z","shell.execute_reply.started":"2025-07-10T16:47:48.173229Z","shell.execute_reply":"2025-07-10T16:48:06.770119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_post(persona, context, max_length=100):\n    input_prompt = f\"[Persona: {persona}] [Context: {context}] =>\"\n    inputs = tokenizer(input_prompt, return_tensors=\"pt\").input_ids\n    output = model.generate(\n        inputs, \n        max_length=max_length, \n        num_return_sequences=1, \n        do_sample=True, \n        pad_token_id=tokenizer.eos_token_id,\n        eos_token_id=tokenizer.eos_token_id\n    )\n    generated = tokenizer.decode(output[0], skip_special_tokens=True)\n    \n    # Remove the prompt from output to get only the generated post\n    return generated.replace(input_prompt, \"\").strip()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:48:06.771979Z","iopub.execute_input":"2025-07-10T16:48:06.772320Z","iopub.status.idle":"2025-07-10T16:48:06.778458Z","shell.execute_reply.started":"2025-07-10T16:48:06.772290Z","shell.execute_reply":"2025-07-10T16:48:06.777558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(generate_post(\"Psychologist\", \"Feeling lonely during lockdown\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:48:06.779687Z","iopub.execute_input":"2025-07-10T16:48:06.780025Z","iopub.status.idle":"2025-07-10T16:48:09.043770Z","shell.execute_reply.started":"2025-07-10T16:48:06.779994Z","shell.execute_reply":"2025-07-10T16:48:09.042862Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"fine tuning ","metadata":{}},{"cell_type":"code","source":"# 📦 Install if needed\n# !pip install datasets transformers --quiet\n\n# 📚 Imports\nfrom datasets import load_dataset\nfrom transformers import (\n    AutoTokenizer,\n    AutoModelForCausalLM,\n    Trainer,\n    TrainingArguments,\n    DataCollatorForLanguageModeling\n)\nimport re\n\n# 🧼 Clean text function\ndef clean_text(text):\n    if text is None:\n        return \"\"\n    text = re.sub(r\"<.*?>\", \"\", text)\n    text = re.sub(r\"http\\S+\", \"\", text)\n    text = re.sub(r\"\\s+\", \" \", text)\n    return text.strip()\n\n# ✏️ Format prompt function\ndef format_prompt(example):\n    cleaned_post = clean_text(example.get(\"post_text\", \"\"))\n    return {\n        \"prompt\": f\"Persona: {example.get('persona_label', '')}\\nContext: {example.get('history_text', '')}\\nPost:\",\n        \"post_text\": cleaned_post\n    }\n\n# 🔢 Load and sample dataset (dev mode = 50k)\nraw_dataset = load_dataset(\"csv\", data_files=\"final_labels_with_history.csv\")[\"train\"]\nraw_dataset = raw_dataset.shuffle(seed=42).select(range(500))  # 👈 Sample for dev mode\n\n# 🧼 Format prompts (multi-processing)\nformatted_dataset = raw_dataset.map(format_prompt, num_proc=4, desc=\"Formatting\")\n\n# 🔠 Load tokenizer\ntokenizer = AutoTokenizer.from_pretrained(\"distilgpt2\")\ntokenizer.pad_token = tokenizer.eos_token\n\n# ✂️ Tokenization function\ndef tokenize_fn(examples):\n    full_texts = [p + \" \" + t for p, t in zip(examples[\"prompt\"], examples[\"post_text\"])]\n    tokenized = tokenizer(\n        full_texts,\n        truncation=True,\n        padding=\"max_length\",\n        max_length=64  # 👈 shorter seq length for speed\n    )\n    tokenized[\"labels\"] = tokenized[\"input_ids\"].copy()\n    return tokenized\n\n# 🔄 Tokenize dataset (batched + parallel)\ntokenized_dataset = formatted_dataset.map(tokenize_fn, batched=True, num_proc=4, desc=\"Tokenizing\")\n\n# 🧹 Keep only necessary columns\ncolumns_to_keep = [\"input_ids\", \"attention_mask\", \"labels\"]\ntokenized_dataset = tokenized_dataset.remove_columns([col for col in tokenized_dataset.column_names if col not in columns_to_keep])\n\n# 🤖 Load model\nmodel = AutoModelForCausalLM.from_pretrained(\"distilgpt2\")\n\n# 📦 Data collator\ndata_collator = DataCollatorForLanguageModeling(tokenizer=tokenizer, mlm=False)\n\n# ⚙️ Training arguments\ntraining_args = TrainingArguments(\n    output_dir=\"./distilgpt2-socialsim\",\n    overwrite_output_dir=True,\n    per_device_train_batch_size=4,\n    num_train_epochs=3,\n    logging_steps=10,\n    save_steps=100,\n    save_total_limit=2,\n    fp16=True,\n    report_to=[],\n    run_name=\"socialsim-gpt2-dev\"\n)\n\n# 🚂 Trainer\ntrainer = Trainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_dataset,\n    data_collator=data_collator,\n)\n\n# 🚀 Train the model\ntrainer.train()\n\n# ✨ Inference function (optimized)\ndef generate_post(persona, context, max_length=100):\n    input_prompt = f\"Persona: {persona}\\nContext: {context}\\nPost:\"\n    inputs = tokenizer(input_prompt, return_tensors=\"pt\").input_ids\n    output = model.generate(\n        inputs,\n        max_length=max_length,\n        do_sample=True,\n        top_p=0.9,\n        temperature=0.8,\n        repetition_penalty=1.1,\n        pad_token_id=tokenizer.eos_token_id,\n        eos_token_id=tokenizer.eos_token_id\n    )\n    generated_text = tokenizer.decode(output[0], skip_special_tokens=True)\n    return generated_text.replace(input_prompt, \"\").strip()\n\n# 🔍 Try a generation\nprint(generate_post(\"Psychologist\", \"Feeling lonely during lockdown\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:48:09.044760Z","iopub.execute_input":"2025-07-10T16:48:09.045032Z","iopub.status.idle":"2025-07-10T16:57:18.041464Z","shell.execute_reply.started":"2025-07-10T16:48:09.045007Z","shell.execute_reply":"2025-07-10T16:57:18.040478Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"final ","metadata":{}},{"cell_type":"code","source":"import os\nprint(os.listdir(\"/kaggle/input/social-sim-challenge-social-media-based-personas/test\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:57:18.042541Z","iopub.execute_input":"2025-07-10T16:57:18.042804Z","iopub.status.idle":"2025-07-10T16:57:18.047833Z","shell.execute_reply.started":"2025-07-10T16:57:18.042783Z","shell.execute_reply":"2025-07-10T16:57:18.046955Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport json\nimport pandas as pd\n\n# ✅ 1. Define the test directory path\ntest_dir = '/kaggle/input/social-sim-challenge-social-media-based-personas/test/'\n\n# ✅ 2. Collect all .jsonl file paths\njsonl_files = [os.path.join(test_dir, f) for f in os.listdir(test_dir) if f.endswith('.jsonl')]\n\n# ✅ 3. Load all threads into a list\nrows = []\nfor file in jsonl_files:\n    with open(file, 'r') as f:\n        for line in f:\n            row = json.loads(line)\n            rows.append({\n                \"id\": row.get(\"id\"),\n                \"cluster_id\": row.get(\"cluster_id\"),\n                \"persona_label\": row.get(\"persona\", {}).get(\"persona_label\", \"User\"),\n                \"history_text\": row.get(\"thread\", [])\n            })\n\n# ✅ 4. Convert to DataFrame\ntest_df = pd.DataFrame(rows)\n\n# ✅ 5. Flatten thread history to plain text (used for generation/classification)\ndef flatten_thread(thread):\n    return \" \".join([t[\"text\"] for t in thread if \"text\" in t])\n\ntest_df[\"history_text\"] = test_df[\"history_text\"].apply(flatten_thread)\n\n# ✅ 6. Final Preview\ntest_df.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:57:18.048699Z","iopub.execute_input":"2025-07-10T16:57:18.048976Z","iopub.status.idle":"2025-07-10T16:57:23.461753Z","shell.execute_reply.started":"2025-07-10T16:57:18.048948Z","shell.execute_reply":"2025-07-10T16:57:23.460680Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nprint(os.listdir(\"/kaggle/working/\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:57:23.462768Z","iopub.execute_input":"2025-07-10T16:57:23.463122Z","iopub.status.idle":"2025-07-10T16:57:23.468858Z","shell.execute_reply.started":"2025-07-10T16:57:23.463089Z","shell.execute_reply":"2025-07-10T16:57:23.467927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sentence_transformers import SentenceTransformer\nfrom sklearn.multioutput import MultiOutputClassifier\nfrom sklearn.linear_model import LogisticRegression\nimport joblib\nimport numpy as np\n\n# 1. Extract texts and labels\ntexts = df_all['history_text'].tolist()\nlabel_columns = [\n    'label_like', 'label_unlike', 'label_repost', 'label_unrepost',\n    'label_follow', 'label_unfollow', 'label_block', 'label_unblock',\n    'label_post_update', 'label_post_delete', 'label_quote', 'label_post',\n    'label_reply'\n]\nlabels = df_all[label_columns].values\n\n# 2. Subsample first 5000 examples\ntexts_small = texts[:5000]\nlabels_small = labels[:5000]\n\n# 3. Filter label columns with only one class\nlabel_columns_filtered = [\n    label_columns[i]\n    for i in range(labels_small.shape[1])\n    if len(np.unique(labels_small[:, i])) > 1\n]\nlabels_small_filtered = df_all[label_columns_filtered].values[:5000]\n\n# 4. Embed texts\nembedder = SentenceTransformer('all-MiniLM-L6-v2')\nX_small = embedder.encode(\n    texts_small,\n    batch_size=512,\n    convert_to_numpy=True,\n    show_progress_bar=True\n)\n\n# 5. Train classifier\nclf = MultiOutputClassifier(LogisticRegression(max_iter=1000))\nclf.fit(X_small, labels_small_filtered)\n\n# 6. Save model\njoblib.dump(clf, \"classifier_small.pkl\")\nprint(\"✅ Model trained and saved as classifier_small.pkl\")\nprint(\"✅ Used labels:\", label_columns_filtered)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:57:23.470075Z","iopub.execute_input":"2025-07-10T16:57:23.470908Z","iopub.status.idle":"2025-07-10T16:59:04.832306Z","shell.execute_reply.started":"2025-07-10T16:57:23.470881Z","shell.execute_reply":"2025-07-10T16:59:04.831397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\nfrom transformers import GPT2Tokenizer, GPT2LMHeadModel\nimport torch\n\n# ✅ Auto-detect device\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\n# Load classifier\nclf = joblib.load(\"/kaggle/working/classifier_small.pkl\")  # your local path\n\n# Load GPT2 generator\ntokenizer_gpt2 = GPT2Tokenizer.from_pretrained(\"distilgpt2\")\nmodel_gpt2 = GPT2LMHeadModel.from_pretrained(\"distilgpt2\").to(device)\nmodel_gpt2.eval()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T16:59:04.833307Z","iopub.execute_input":"2025-07-10T16:59:04.833608Z","iopub.status.idle":"2025-07-10T16:59:05.270351Z","shell.execute_reply.started":"2025-07-10T16:59:04.833585Z","shell.execute_reply":"2025-07-10T16:59:05.269564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport json\n\n# Path to the test folder\ntest_dir = \"/kaggle/input/social-sim-challenge-social-media-based-personas/test/\"\n\n# List all .jsonl files in the folder\nfiles = [f for f in os.listdir(test_dir) if f.endswith(\".jsonl\")]\n\n# Parse them all into one DataFrame\ntest_df_list = []\nfor file in files:\n    with open(os.path.join(test_dir, file), 'r') as f:\n        for line in f:\n            test_df_list.append(json.loads(line))\n\ntest_df = pd.DataFrame(test_df_list)\nprint(\"✅ Combined test rows:\", test_df.shape[0])\nprint(\"✅ Columns:\", test_df.columns.tolist())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:11:04.899648Z","iopub.execute_input":"2025-07-10T18:11:04.899965Z","iopub.status.idle":"2025-07-10T18:11:08.109376Z","shell.execute_reply.started":"2025-07-10T18:11:04.899944Z","shell.execute_reply":"2025-07-10T18:11:08.108477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom transformers import GPT2LMHeadModel, GPT2Tokenizer\nimport pandas as pd\nfrom tqdm import tqdm\nimport os\n\n# === Setup ===\ntokenizer = GPT2Tokenizer.from_pretrained(\"distilgpt2\")\ntokenizer.pad_token = tokenizer.eos_token  # Fix padding issue\ntokenizer.padding_side = \"left\"\nmodel = GPT2LMHeadModel.from_pretrained(\"distilgpt2\")\nmodel.eval()\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel = model.to(device)\n\n# === Load your test data ===\ntest_df = pd.read_csv(\"/kaggle/working/final_labels_with_history.csv\")\n\n# === Select action ===\nactions = [\"label_post\"]  # You can extend this list later\nbatch_size = 32\nmax_rows = 10000  # Use top 10k rows for quick testing\n\nfor act in actions:\n    print(f\"\\nProcessing: {act}\")\n\n    # Create output column with default 'EMPTY'\n    test_df[act] = \"EMPTY\"\n\n    # Filter only True rows\n    active_rows = test_df[test_df[act] == True].head(max_rows).copy()\n\n    # Store generated results\n    generated_texts = []\n    indices = active_rows.index.tolist()\n    texts_to_generate = active_rows[\"history_text\"].tolist()\n\n    # === Generation Function ===\n    def generate_texts(texts, max_length=130):\n        inputs = tokenizer(\n            texts,\n            return_tensors=\"pt\",\n            padding=True,\n            truncation=True,\n            max_length=max_length\n        ).to(device)\n        with torch.no_grad():\n            outputs = model.generate(\n                **inputs,\n                max_length=max_length,\n                do_sample=True,\n                top_k=50,\n                top_p=0.95,\n                temperature=0.7,\n                pad_token_id=tokenizer.eos_token_id\n            )\n        return tokenizer.batch_decode(outputs, skip_special_tokens=True)\n\n    # Generate in batches\n    for i in tqdm(range(0, len(texts_to_generate), batch_size), desc=f\"Generating for {act}\"):\n        batch_texts = texts_to_generate[i:i + batch_size]\n        batch_outputs = generate_texts(batch_texts)\n        generated_texts.extend(batch_outputs)\n\n    # Update test_df with generated texts\n    for idx, gen_text in zip(indices, generated_texts):\n        test_df.at[idx, act] = gen_text\n\n    # Save partial result\n    part_file = f\"submission_{act}_part.csv\"\n    test_df[[\"id\", \"cluster_id\", act]].to_csv(part_file, index=False)\n    print(f\"Saved partial result to: {part_file}\")\n\n# === Final Merge ===\nprint(\"\\nMerging final submission file...\")\nfinal_df = test_df[[\"id\", \"cluster_id\"] + actions]\nfinal_df.to_csv(\"final_submission.csv\", index=False)\nprint(\"Final submission saved as 'final_submission.csv'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:11:29.454769Z","iopub.execute_input":"2025-07-10T18:11:29.455072Z","iopub.status.idle":"2025-07-10T18:12:49.168331Z","shell.execute_reply.started":"2025-07-10T18:11:29.455051Z","shell.execute_reply":"2025-07-10T18:12:49.167385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('final_submission.csv')\n\nprint(df.shape)  # Should match test dataset rows\n\nprint(df.columns)  # Check columns\n\nprint(df.head())  # Sample data\n\n# Check for missing or null values\nprint(df.isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:13:55.394213Z","iopub.execute_input":"2025-07-10T18:13:55.394648Z","iopub.status.idle":"2025-07-10T18:14:03.213471Z","shell.execute_reply.started":"2025-07-10T18:13:55.394614Z","shell.execute_reply":"2025-07-10T18:14:03.212591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read the file line by line to inspect its raw content\nsubmission_filename = 'submission.csv'\n\nprint(f\"\\n--- Raw content of '{submission_filename}' (first 10 lines) ---\")\ntry:\n    with open(submission_filename, 'r') as f:\n        for i, line in enumerate(f):\n            print(line.strip()) # .strip() removes newline characters\n            if i >= 9: # Print only first 10 lines (header + 9 data rows)\n                break\nexcept FileNotFoundError:\n    print(f\"Error: The file '{submission_filename}' was not found.\")\n    print(\"Please ensure you have run the code to generate the submission.csv file in your notebook environment first.\")\nexcept Exception as e:\n    print(f\"An error occurred while reading the file: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:24:28.914724Z","iopub.execute_input":"2025-07-10T18:24:28.915092Z","iopub.status.idle":"2025-07-10T18:24:28.922022Z","shell.execute_reply.started":"2025-07-10T18:24:28.915067Z","shell.execute_reply":"2025-07-10T18:24:28.921140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv # Import the csv module at the top of your script\n\n# ... (your existing code to create final_submission_df,\n# including the .apply(lambda x: \"True\" if x else \"False\") for action columns) ...\n\n# Save the DataFrame to a CSV file for submission\noutput_filename = 'submission.csv'\nfinal_submission_df.to_csv(output_filename, index=False, quoting=csv.QUOTE_NONNUMERIC)\nprint(f\"\\nFinal submission saved as '{output_filename}' with proper quoting.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:24:44.459762Z","iopub.execute_input":"2025-07-10T18:24:44.460117Z","iopub.status.idle":"2025-07-10T18:24:44.469036Z","shell.execute_reply.started":"2025-07-10T18:24:44.460091Z","shell.execute_reply":"2025-07-10T18:24:44.468220Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import csv # Make sure this is at the top of your script\n\n# ... (your code to create final_submission_df) ...\n\nfinal_submission_df.to_csv(output_filename, index=False, quoting=csv.QUOTE_NONNUMERIC)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T18:25:03.829564Z","iopub.execute_input":"2025-07-10T18:25:03.830272Z","iopub.status.idle":"2025-07-10T18:25:03.836996Z","shell.execute_reply.started":"2025-07-10T18:25:03.830243Z","shell.execute_reply":"2025-07-10T18:25:03.835949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}