{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.12"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":19018,"databundleVersionId":2703900},{"sourceType":"datasetVersion","sourceId":3053020,"datasetId":1869236,"databundleVersionId":3101552},{"sourceType":"datasetVersion","sourceId":1257215,"datasetId":723100,"databundleVersionId":1289103},{"sourceType":"datasetVersion","sourceId":2798066,"datasetId":1709138,"databundleVersionId":2844101},{"sourceType":"datasetVersion","sourceId":16197047,"datasetId":10384923,"databundleVersionId":17175904}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Advanced Cyberbullying Detection Using Machine Learning\n\nComplete project notebook with:\n- Multi-dataset training\n- Multilingual toxic comments\n- TF-IDF vectorization\n- Multiple ML models\n- GPU XGBoost\n- Automatic best model selection\n- Advanced evaluation graphs\n- Real-time prediction\n","metadata":{}},{"cell_type":"code","source":"!pip install nltk xgboost scikit-learn pandas numpy matplotlib seaborn wordcloud imbalanced-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:22.752633Z","iopub.execute_input":"2026-05-15T05:49:22.752947Z","iopub.status.idle":"2026-05-15T05:49:27.312748Z","shell.execute_reply.started":"2026-05-15T05:49:22.752913Z","shell.execute_reply":"2026-05-15T05:49:27.311795Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport nltk\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom collections import Counter\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    confusion_matrix,\n    classification_report,\n    roc_curve,\n    auc\n)\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import LinearSVC\n\nfrom xgboost import XGBClassifier\n\nfrom wordcloud import WordCloud\n\nnltk.download('stopwords')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:27.314638Z","iopub.execute_input":"2026-05-15T05:49:27.315487Z","iopub.status.idle":"2026-05-15T05:49:30.227224Z","shell.execute_reply.started":"2026-05-15T05:49:27.315452Z","shell.execute_reply":"2026-05-15T05:49:30.226574Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Datasets","metadata":{}},{"cell_type":"code","source":"path1 = '/kaggle/input/datasets/andrewmvd/cyberbullying-classification/cyberbullying_tweets.csv'\n\npath2 = '/kaggle/input/datasets/mrmorj/hate-speech-and-offensive-language-dataset/labeled_data.csv'\n\npath3 = '/kaggle/input/datasets/julian3833/jigsaw-toxic-comment-classification-challenge/train.csv'\n\npath4 = '/kaggle/input/competitions/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv'\n\ndf1 = pd.read_csv(path1)\ndf2 = pd.read_csv(path2)\ndf3 = pd.read_csv(path3)\ndf4 = pd.read_csv(path4)\n\nprint(df1.shape)\nprint(df2.shape)\nprint(df3.shape)\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:30.228247Z","iopub.execute_input":"2026-05-15T05:49:30.228681Z","iopub.status.idle":"2026-05-15T05:49:34.415121Z","shell.execute_reply.started":"2026-05-15T05:49:30.228655Z","shell.execute_reply":"2026-05-15T05:49:34.414245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Advanced Text Preprocessing\n\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\nnltk.download('stopwords')\nnltk.download('wordnet')\nnltk.download('omw-1.4')\n\nstop_words = set(stopwords.words('english'))\n\nlemmatizer = WordNetLemmatizer()\n\ndef advanced_clean_text(text):\n\n    text = str(text).lower()\n\n    # Remove URLs\n    text = re.sub(r'http\\S+|www\\S+|https\\S+', '', text)\n\n    # Remove mentions and hashtags\n    text = re.sub(r'@\\w+|#\\w+', '', text)\n\n    # Remove numbers\n    text = re.sub(r'\\d+', '', text)\n\n    # Remove punctuation\n    text = re.sub(r'[^a-zA-Z\\s]', '', text)\n\n    # Remove extra spaces\n    text = re.sub(r'\\s+', ' ', text).strip()\n\n    words = text.split()\n\n    # Remove stopwords and lemmatize\n    words = [\n        lemmatizer.lemmatize(word)\n        for word in words\n        if word not in stop_words and len(word) > 2\n    ]\n\n    return ' '.join(words)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:34.416215Z","iopub.execute_input":"2026-05-15T05:49:34.416627Z","iopub.status.idle":"2026-05-15T05:49:34.669826Z","shell.execute_reply.started":"2026-05-15T05:49:34.416593Z","shell.execute_reply":"2026-05-15T05:49:34.668981Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Preprocessing","metadata":{}},{"cell_type":"code","source":"# DATASET 1\ndf1 = df1.rename(columns={\n    'tweet_text':'text',\n    'cyberbullying_type':'label'\n})\n\ndf1 = df1[['text', 'label']]\n\ndf1['label'] = df1['label'].apply(\n    lambda x: 0 if x == 'not_cyberbullying' else 1\n)\n\n# DATASET 2\ndf2 = df2.rename(columns={\n    'tweet':'text',\n    'class':'label'\n})\n\ndf2 = df2[['text', 'label']]\n\ndf2['label'] = df2['label'].apply(\n    lambda x: 0 if x == 2 else 1\n)\n\n# DATASET 3\ndf3['label'] = (\n    df3[['toxic', 'severe_toxic',\n         'obscene', 'threat',\n         'insult', 'identity_hate']]\n    .sum(axis=1)\n)\n\ndf3['label'] = df3['label'].apply(\n    lambda x: 1 if x > 0 else 0\n)\n\ndf3 = df3.rename(columns={\n    'comment_text':'text'\n})\n\ndf3 = df3[['text', 'label']]\n\n# DATASET 4\ndf4 = df4.rename(columns={\n    'comment_text':'text',\n    'toxic':'label'\n})\n\ndf4 = df4[['text', 'label']]\n\ndf4.dropna(inplace=True)\n\n# Sample multilingual dataset for RAM optimization\ndf4 = df4.sample(\n    n=100000,\n    random_state=42\n)\n\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:34.672027Z","iopub.execute_input":"2026-05-15T05:49:34.672550Z","iopub.status.idle":"2026-05-15T05:49:34.891622Z","shell.execute_reply.started":"2026-05-15T05:49:34.672524Z","shell.execute_reply":"2026-05-15T05:49:34.890633Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Merge Datasets","metadata":{}},{"cell_type":"code","source":"combined_df = pd.concat(\n    [df1, df2, df3, df4],\n    ignore_index=True\n)\n\ncombined_df.drop_duplicates(inplace=True)\ncombined_df.dropna(inplace=True)\n\ncombined_df['label'] = combined_df['label'].astype(int)\n\ncombined_df = combined_df[\n    combined_df['label'].isin([0,1])\n]\n\nprint(combined_df.shape)\n\nprint(combined_df['label'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:34.892910Z","iopub.execute_input":"2026-05-15T05:49:34.893533Z","iopub.status.idle":"2026-05-15T05:49:35.334230Z","shell.execute_reply.started":"2026-05-15T05:49:34.893391Z","shell.execute_reply":"2026-05-15T05:49:35.333289Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Label Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6,5))\n\nsns.countplot(\n    x=combined_df['label']\n)\n\nplt.title('Cyberbullying vs Non-Cyberbullying')\n\nplt.xticks(\n    [0,1],\n    ['Non-Cyberbullying', 'Cyberbullying']\n)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:35.335256Z","iopub.execute_input":"2026-05-15T05:49:35.335629Z","iopub.status.idle":"2026-05-15T05:49:35.866459Z","shell.execute_reply.started":"2026-05-15T05:49:35.335602Z","shell.execute_reply":"2026-05-15T05:49:35.865714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Text Cleaning","metadata":{}},{"cell_type":"code","source":"combined_df['clean_text'] = combined_df['text'].apply(advanced_clean_text)\n\nprint(\"Sample Cleaned Text:\")\nprint(combined_df[['text', 'clean_text']].head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:49:35.867326Z","iopub.execute_input":"2026-05-15T05:49:35.867634Z","iopub.status.idle":"2026-05-15T05:50:16.385789Z","shell.execute_reply.started":"2026-05-15T05:49:35.867610Z","shell.execute_reply":"2026-05-15T05:50:16.385009Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## WordCloud","metadata":{}},{"cell_type":"code","source":"text_data = ' '.join(combined_df['clean_text'])\n\nwordcloud = WordCloud(\n    width=1200,\n    height=600,\n    background_color='white'\n).generate(text_data)\n\nplt.figure(figsize=(14,7))\n\nplt.imshow(wordcloud)\n\nplt.axis('off')\n\nplt.title('Cyberbullying WordCloud')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:50:16.386947Z","iopub.execute_input":"2026-05-15T05:50:16.387311Z","iopub.status.idle":"2026-05-15T05:50:53.709884Z","shell.execute_reply.started":"2026-05-15T05:50:16.387284Z","shell.execute_reply":"2026-05-15T05:50:53.708913Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TF-IDF Vectorization","metadata":{}},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(\n    max_features=80000,\n    ngram_range=(1,2),\n    min_df=3,\n    max_df=0.9,\n    sublinear_tf=True,\n    strip_accents='unicode',\n    dtype=np.float32\n)\n\nX = vectorizer.fit_transform(\n    combined_df['clean_text']\n)\n\ny = combined_df['label']\n\nprint(X.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:50:53.711674Z","iopub.execute_input":"2026-05-15T05:50:53.712173Z","iopub.status.idle":"2026-05-15T05:51:19.730787Z","shell.execute_reply.started":"2026-05-15T05:50:53.712137Z","shell.execute_reply":"2026-05-15T05:51:19.730087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train-Test Split","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X,\n    y,\n    test_size=0.2,\n    random_state=42,\n    stratify=y\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:51:19.731685Z","iopub.execute_input":"2026-05-15T05:51:19.732004Z","iopub.status.idle":"2026-05-15T05:51:19.850395Z","shell.execute_reply.started":"2026-05-15T05:51:19.731979Z","shell.execute_reply":"2026-05-15T05:51:19.849505Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handle Class Imbalance","metadata":{}},{"cell_type":"code","source":"counter = Counter(y)\n\nscale_pos_weight = counter[0] / counter[1]\n\nprint(counter)\n\nprint(scale_pos_weight)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:51:19.851462Z","iopub.execute_input":"2026-05-15T05:51:19.851684Z","iopub.status.idle":"2026-05-15T05:51:19.893158Z","shell.execute_reply.started":"2026-05-15T05:51:19.851661Z","shell.execute_reply":"2026-05-15T05:51:19.892235Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Models","metadata":{}},{"cell_type":"code","source":"# Logistic Regression\nlr_model = LogisticRegression(\n    max_iter=1000\n)\n\nlr_model.fit(X_train, y_train)\n\nlr_pred = lr_model.predict(X_test)\n\n# Naive Bayes\nnb_model = MultinomialNB()\n\nnb_model.fit(X_train, y_train)\n\nnb_pred = nb_model.predict(X_test)\n\n# Random Forest\nrf_model = RandomForestClassifier(\n\n    n_estimators=50,\n\n    max_depth=20,\n\n    n_jobs=-1,\n\n    random_state=42\n)\n\nrf_model.fit(X_train, y_train)\n\nrf_pred = rf_model.predict(X_test)\n\n# SVM\nsvm_model = LinearSVC()\n\nsvm_model.fit(X_train, y_train)\n\nsvm_pred = svm_model.predict(X_test)\n\n# GPU XGBoost\nxgb_model = XGBClassifier(\n\n    tree_method='hist',\n    device='cuda',\n\n    n_estimators=1000,\n\n    learning_rate=0.03,\n\n    max_depth=12,\n\n    min_child_weight=2,\n\n    subsample=0.85,\n\n    colsample_bytree=0.85,\n\n    gamma=0.15,\n\n    scale_pos_weight=scale_pos_weight,\n\n    eval_metric='logloss',\n\n    random_state=42,\n\n    n_jobs=-1\n)\n\nxgb_model.fit(\n    X_train,\n    y_train\n)\n\ny_probs = xgb_model.predict_proba(X_test)[:,1]\n\nxgb_pred = (y_probs > 0.35).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:51:19.894362Z","iopub.execute_input":"2026-05-15T05:51:19.894767Z","iopub.status.idle":"2026-05-15T05:52:17.160823Z","shell.execute_reply.started":"2026-05-15T05:51:19.894729Z","shell.execute_reply":"2026-05-15T05:52:17.159749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Auto Select Best Model","metadata":{}},{"cell_type":"code","source":"models = {\n    'Logistic Regression': lr_model,\n    'Naive Bayes': nb_model,\n    'Random Forest': rf_model,\n    'SVM': svm_model,\n    'XGBoost': xgb_model\n}\n\npredictions = {\n    'Logistic Regression': lr_pred,\n    'Naive Bayes': nb_pred,\n    'Random Forest': rf_pred,\n    'SVM': svm_pred,\n    'XGBoost': xgb_pred\n}\n\nmodel_scores = {}\n\nfor name in models:\n\n    score = f1_score(\n        y_test,\n        predictions[name]\n    )\n\n    model_scores[name] = score\n\nbest_model_name = max(\n    model_scores,\n    key=model_scores.get\n)\n\nbest_model = models[best_model_name]\n\nprint('Best Model:', best_model_name)\n\nprint('Best F1 Score:', model_scores[best_model_name])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.163546Z","iopub.execute_input":"2026-05-15T05:52:17.163880Z","iopub.status.idle":"2026-05-15T05:52:17.198762Z","shell.execute_reply.started":"2026-05-15T05:52:17.163851Z","shell.execute_reply":"2026-05-15T05:52:17.197929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Classification Reports","metadata":{}},{"cell_type":"code","source":"print('LOGISTIC REGRESSION')\nprint(classification_report(y_test, lr_pred))\n\nprint('NAIVE BAYES')\nprint(classification_report(y_test, nb_pred))\n\nprint('RANDOM FOREST')\nprint(classification_report(y_test, rf_pred))\n\nprint('SVM')\nprint(classification_report(y_test, svm_pred))\n\nprint('XGBOOST')\nprint(classification_report(y_test, xgb_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.199866Z","iopub.execute_input":"2026-05-15T05:52:17.200220Z","iopub.status.idle":"2026-05-15T05:52:17.287687Z","shell.execute_reply.started":"2026-05-15T05:52:17.200184Z","shell.execute_reply":"2026-05-15T05:52:17.286947Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Metrics Comparison","metadata":{}},{"cell_type":"code","source":"metrics_df = pd.DataFrame({\n\n    'Model': [\n        'Logistic Regression',\n        'Naive Bayes',\n        'Random Forest',\n        'SVM',\n        'XGBoost'\n    ],\n\n    'Accuracy': [\n        accuracy_score(y_test, lr_pred),\n        accuracy_score(y_test, nb_pred),\n        accuracy_score(y_test, rf_pred),\n        accuracy_score(y_test, svm_pred),\n        accuracy_score(y_test, xgb_pred)\n    ],\n\n    'Precision': [\n        precision_score(y_test, lr_pred),\n        precision_score(y_test, nb_pred),\n        precision_score(y_test, rf_pred),\n        precision_score(y_test, svm_pred),\n        precision_score(y_test, xgb_pred)\n    ],\n\n    'Recall': [\n        recall_score(y_test, lr_pred),\n        recall_score(y_test, nb_pred),\n        recall_score(y_test, rf_pred),\n        recall_score(y_test, svm_pred),\n        recall_score(y_test, xgb_pred)\n    ],\n\n    'F1-Score': [\n        f1_score(y_test, lr_pred),\n        f1_score(y_test, nb_pred),\n        f1_score(y_test, rf_pred),\n        f1_score(y_test, svm_pred),\n        f1_score(y_test, xgb_pred)\n    ]\n})\n\nmetrics_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.288669Z","iopub.execute_input":"2026-05-15T05:52:17.289031Z","iopub.status.idle":"2026-05-15T05:52:17.391007Z","shell.execute_reply.started":"2026-05-15T05:52:17.288994Z","shell.execute_reply":"2026-05-15T05:52:17.390271Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Accuracy Comparison Graph","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\nsns.barplot(\n    x='Model',\n    y='Accuracy',\n    data=metrics_df\n)\n\nplt.title('Accuracy Comparison of ML Models')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.391979Z","iopub.execute_input":"2026-05-15T05:52:17.392907Z","iopub.status.idle":"2026-05-15T05:52:17.559027Z","shell.execute_reply.started":"2026-05-15T05:52:17.392860Z","shell.execute_reply":"2026-05-15T05:52:17.558036Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Precision Recall F1 Comparison","metadata":{}},{"cell_type":"code","source":"metrics_melted = metrics_df.melt(\n    id_vars='Model',\n    value_vars=['Precision', 'Recall', 'F1-Score']\n)\n\nplt.figure(figsize=(12,6))\n\nsns.barplot(\n    x='Model',\n    y='value',\n    hue='variable',\n    data=metrics_melted\n)\n\nplt.title('Precision Recall and F1 Score Comparison')\n\nplt.ylabel('Score')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.560036Z","iopub.execute_input":"2026-05-15T05:52:17.560418Z","iopub.status.idle":"2026-05-15T05:52:17.773762Z","shell.execute_reply.started":"2026-05-15T05:52:17.560391Z","shell.execute_reply":"2026-05-15T05:52:17.773096Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"# Confusion Matrix for Best Model\n\nbest_pred = predictions[best_model_name]\n\ncm = confusion_matrix(y_test, best_pred)\n\nplt.figure(figsize=(7,6))\n\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt='d',\n    cmap='Greens'\n)\n\nplt.title(f'Confusion Matrix - {best_model_name}')\n\nplt.xlabel('Predicted')\n\nplt.ylabel('Actual')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.774902Z","iopub.execute_input":"2026-05-15T05:52:17.775258Z","iopub.status.idle":"2026-05-15T05:52:17.956702Z","shell.execute_reply.started":"2026-05-15T05:52:17.775232Z","shell.execute_reply":"2026-05-15T05:52:17.955869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"cell_type":"code","source":"# ROC Curve for Best Model\n\nif hasattr(best_model, 'predict_proba'):\n    \n    best_probs = best_model.predict_proba(X_test)[:,1]\n\nelse:\n    \n    best_scores = best_model.decision_function(X_test)\n    \n    best_probs = (\n        best_scores - best_scores.min()\n    ) / (\n        best_scores.max() - best_scores.min()\n    )\n\nfpr, tpr, thresholds = roc_curve(\n    y_test,\n    best_probs\n)\n\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7,6))\n\nplt.plot(\n    fpr,\n    tpr,\n    linewidth=2,\n    label=f'AUC = {roc_auc:.3f}'\n)\n\nplt.plot([0,1], [0,1], linestyle='--')\n\nplt.xlabel('False Positive Rate')\n\nplt.ylabel('True Positive Rate')\n\nplt.title(f'ROC Curve - {best_model_name}')\n\nplt.legend()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:17.957600Z","iopub.execute_input":"2026-05-15T05:52:17.957984Z","iopub.status.idle":"2026-05-15T05:52:18.112731Z","shell.execute_reply.started":"2026-05-15T05:52:17.957956Z","shell.execute_reply":"2026-05-15T05:52:18.112024Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"# Feature Importance for Best Tree-Based Model\n\ntree_model = xgb_model if best_model_name == 'XGBoost' else rf_model\n\nimportance = tree_model.feature_importances_\n\nfeature_names = vectorizer.get_feature_names_out()\n\nfeature_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': importance\n})\n\nfeature_df = feature_df.sort_values(\n    by='Importance',\n    ascending=False\n).head(20)\n\nplt.figure(figsize=(12,8))\n\nsns.barplot(\n    x='Importance',\n    y='Feature',\n    data=feature_df\n)\n\nplt.title(f'Top 20 Important Features - {type(tree_model).__name__}')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:18.113595Z","iopub.execute_input":"2026-05-15T05:52:18.113933Z","iopub.status.idle":"2026-05-15T05:52:18.516442Z","shell.execute_reply.started":"2026-05-15T05:52:18.113904Z","shell.execute_reply":"2026-05-15T05:52:18.515759Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Additional Relevant Graphs","metadata":{}},{"cell_type":"code","source":"# Additional Relevant Graphs\n\n# Model Comparison Heatmap\nplt.figure(figsize=(8,5))\n\nheatmap_df = metrics_df.set_index('Model')\n\nsns.heatmap(\n    heatmap_df,\n    annot=True,\n    cmap='YlGnBu',\n    fmt='.3f'\n)\n\nplt.title('Model Performance Heatmap')\n\nplt.show()\n\n# Accuracy vs F1 Score\nplt.figure(figsize=(8,5))\n\nplt.plot(\n    metrics_df['Model'],\n    metrics_df['Accuracy'],\n    marker='o',\n    linewidth=2,\n    label='Accuracy'\n)\n\nplt.plot(\n    metrics_df['Model'],\n    metrics_df['F1-Score'],\n    marker='s',\n    linewidth=2,\n    label='F1 Score'\n)\n\nplt.xticks(rotation=15)\n\nplt.ylabel('Score')\n\nplt.title('Accuracy vs F1 Score Comparison')\n\nplt.legend()\n\nplt.grid(True)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:18.517590Z","iopub.execute_input":"2026-05-15T05:52:18.517918Z","iopub.status.idle":"2026-05-15T05:52:18.864050Z","shell.execute_reply.started":"2026-05-15T05:52:18.517879Z","shell.execute_reply":"2026-05-15T05:52:18.863266Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Additional NLP Visualizations","metadata":{}},{"cell_type":"code","source":"# Additional NLP Visualizations\n\nfrom collections import Counter\n\n# Most Common Words\nall_words = ' '.join(combined_df['clean_text']).split()\n\ncommon_words = Counter(all_words).most_common(20)\n\nwords = [word[0] for word in common_words]\n\ncounts = [word[1] for word in common_words]\n\nplt.figure(figsize=(12,6))\n\nplt.bar(words, counts)\n\nplt.xticks(rotation=45)\n\nplt.title('Top 20 Most Common Cleaned Words')\n\nplt.xlabel('Words')\n\nplt.ylabel('Frequency')\n\nplt.show()\n\n# Text Length Distribution\ntext_lengths = combined_df['clean_text'].apply(lambda x: len(x.split()))\n\nplt.figure(figsize=(8,5))\n\nplt.hist(text_lengths, bins=30)\n\nplt.title('Distribution of Text Lengths')\n\nplt.xlabel('Number of Words')\n\nplt.ylabel('Frequency')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:18.865138Z","iopub.execute_input":"2026-05-15T05:52:18.865549Z","iopub.status.idle":"2026-05-15T05:52:21.699553Z","shell.execute_reply.started":"2026-05-15T05:52:18.865506Z","shell.execute_reply":"2026-05-15T05:52:21.698803Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Real Time Prediction","metadata":{}},{"cell_type":"code","source":"def predict_cyberbullying(text):\n\n    cleaned = advanced_clean_text(text)\n\n    vectorized = vectorizer.transform([cleaned])\n\n    if hasattr(best_model, 'predict_proba'):\n\n        probability = best_model.predict_proba(vectorized)[0][1]\n\n        prediction = 1 if probability > 0.35 else 0\n\n        score = probability\n\n    elif hasattr(best_model, 'decision_function'):\n\n        score = best_model.decision_function(vectorized)[0]\n\n        prediction = 1 if score > -0.5 else 0\n\n    else:\n\n        prediction = best_model.predict(vectorized)[0]\n\n        score = 0\n\n    if prediction == 1:\n\n        return (\n            f'Cyberbullying Detected '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\n    else:\n\n        return (\n            f'Non-Cyberbullying '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\nprint(predict_cyberbullying(\"You are useless and nobody likes you\"))\n\nprint(predict_cyberbullying(\"Go kill yourself\"))\n\nprint(predict_cyberbullying(\"Have a nice day\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-15T05:52:21.700564Z","iopub.execute_input":"2026-05-15T05:52:21.700929Z","iopub.status.idle":"2026-05-15T05:52:21.711404Z","shell.execute_reply.started":"2026-05-15T05:52:21.700875Z","shell.execute_reply":"2026-05-15T05:52:21.710516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}