{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":19018,"databundleVersionId":2703900},{"sourceType":"datasetVersion","sourceId":3053020,"datasetId":1869236,"databundleVersionId":3101552},{"sourceType":"datasetVersion","sourceId":1257215,"datasetId":723100,"databundleVersionId":1289103},{"sourceType":"datasetVersion","sourceId":2798066,"datasetId":1709138,"databundleVersionId":2844101},{"sourceType":"datasetVersion","sourceId":16197047,"datasetId":10384923,"databundleVersionId":17175904}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Advanced Cyberbullying Detection Using Machine Learning\n\nComplete project notebook with:\n- Multi-dataset training\n- Multilingual toxic comments\n- TF-IDF vectorization\n- Multiple ML models\n- GPU XGBoost\n- Automatic best model selection\n- Advanced evaluation graphs\n- Real-time prediction\n","metadata":{}},{"cell_type":"code","source":"!pip install nltk xgboost scikit-learn pandas numpy matplotlib seaborn wordcloud imbalanced-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:23:50.063123Z","iopub.execute_input":"2026-05-10T19:23:50.063435Z","iopub.status.idle":"2026-05-10T19:23:56.386714Z","shell.execute_reply.started":"2026-05-10T19:23:50.063397Z","shell.execute_reply":"2026-05-10T19:23:56.385981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport nltk\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom collections import Counter\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    confusion_matrix,\n    classification_report,\n    roc_curve,\n    auc\n)\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import LinearSVC\n\nfrom xgboost import XGBClassifier\n\nfrom wordcloud import WordCloud\n\nnltk.download('stopwords')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:23:56.388952Z","iopub.execute_input":"2026-05-10T19:23:56.389311Z","iopub.status.idle":"2026-05-10T19:24:00.272572Z","shell.execute_reply.started":"2026-05-10T19:23:56.389281Z","shell.execute_reply":"2026-05-10T19:24:00.271901Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load Datasets","metadata":{}},{"cell_type":"code","source":"path1 = '/kaggle/input/datasets/andrewmvd/cyberbullying-classification/cyberbullying_tweets.csv'\n\npath2 = '/kaggle/input/datasets/mrmorj/hate-speech-and-offensive-language-dataset/labeled_data.csv'\n\npath3 = '/kaggle/input/datasets/julian3833/jigsaw-toxic-comment-classification-challenge/train.csv'\n\npath4 = '/kaggle/input/datasets/gjaganmohanachary/multilingual/jigsaw-toxic-comment-train.csv'\n\ndf1 = pd.read_csv(path1)\ndf2 = pd.read_csv(path2)\ndf3 = pd.read_csv(path3)\ndf4 = pd.read_csv(path4)\n\nprint(df1.shape)\nprint(df2.shape)\nprint(df3.shape)\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:27:02.897149Z","iopub.execute_input":"2026-05-10T19:27:02.897368Z","iopub.status.idle":"2026-05-10T19:27:05.808678Z","shell.execute_reply.started":"2026-05-10T19:27:02.897346Z","shell.execute_reply":"2026-05-10T19:27:05.807911Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Dataset Preprocessing","metadata":{}},{"cell_type":"code","source":"# DATASET 1\ndf1 = df1.rename(columns={\n    'tweet_text':'text',\n    'cyberbullying_type':'label'\n})\n\ndf1 = df1[['text', 'label']]\n\ndf1['label'] = df1['label'].apply(\n    lambda x: 0 if x == 'not_cyberbullying' else 1\n)\n\n# DATASET 2\ndf2 = df2.rename(columns={\n    'tweet':'text',\n    'class':'label'\n})\n\ndf2 = df2[['text', 'label']]\n\ndf2['label'] = df2['label'].apply(\n    lambda x: 0 if x == 2 else 1\n)\n\n# DATASET 3\ndf3['label'] = (\n    df3[['toxic', 'severe_toxic',\n         'obscene', 'threat',\n         'insult', 'identity_hate']]\n    .sum(axis=1)\n)\n\ndf3['label'] = df3['label'].apply(\n    lambda x: 1 if x > 0 else 0\n)\n\ndf3 = df3.rename(columns={\n    'comment_text':'text'\n})\n\ndf3 = df3[['text', 'label']]\n\n# DATASET 4\ndf4 = df4.rename(columns={\n    'comment_text':'text',\n    'toxic':'label'\n})\n\ndf4 = df4[['text', 'label']]\n\ndf4.dropna(inplace=True)\n\n# Sample multilingual dataset for RAM optimization\ndf4 = df4.sample(\n    n=100000,\n    random_state=42\n)\n\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:04.027546Z","iopub.execute_input":"2026-05-10T19:24:04.027788Z","iopub.status.idle":"2026-05-10T19:24:04.267298Z","shell.execute_reply.started":"2026-05-10T19:24:04.027764Z","shell.execute_reply":"2026-05-10T19:24:04.266312Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Merge Datasets","metadata":{}},{"cell_type":"code","source":"combined_df = pd.concat(\n    [df1, df2, df3, df4],\n    ignore_index=True\n)\n\ncombined_df.drop_duplicates(inplace=True)\ncombined_df.dropna(inplace=True)\n\ncombined_df['label'] = combined_df['label'].astype(int)\n\ncombined_df = combined_df[\n    combined_df['label'].isin([0,1])\n]\n\nprint(combined_df.shape)\n\nprint(combined_df['label'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:04.268673Z","iopub.execute_input":"2026-05-10T19:24:04.269218Z","iopub.status.idle":"2026-05-10T19:24:04.707207Z","shell.execute_reply.started":"2026-05-10T19:24:04.269084Z","shell.execute_reply":"2026-05-10T19:24:04.706452Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Label Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6,5))\n\nsns.countplot(\n    x=combined_df['label']\n)\n\nplt.title('Cyberbullying vs Non-Cyberbullying')\n\nplt.xticks(\n    [0,1],\n    ['Non-Cyberbullying', 'Cyberbullying']\n)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:04.708209Z","iopub.execute_input":"2026-05-10T19:24:04.708694Z","iopub.status.idle":"2026-05-10T19:24:05.298358Z","shell.execute_reply.started":"2026-05-10T19:24:04.708662Z","shell.execute_reply":"2026-05-10T19:24:05.297484Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Text Cleaning","metadata":{}},{"cell_type":"code","source":"def clean_text(text):\n\n    text = str(text).lower()\n\n    text = re.sub(r'http\\S+', '', text)\n\n    text = re.sub(r'\\n', ' ', text)\n\n    text = re.sub(r'\\s+', ' ', text)\n\n    return text\n\ncombined_df['clean_text'] = combined_df['text'].apply(clean_text)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:05.300700Z","iopub.execute_input":"2026-05-10T19:24:05.301482Z","iopub.status.idle":"2026-05-10T19:24:11.272431Z","shell.execute_reply.started":"2026-05-10T19:24:05.301446Z","shell.execute_reply":"2026-05-10T19:24:11.271865Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## WordCloud","metadata":{}},{"cell_type":"code","source":"text_data = ' '.join(combined_df['clean_text'])\n\nwordcloud = WordCloud(\n    width=1200,\n    height=600,\n    background_color='white'\n).generate(text_data)\n\nplt.figure(figsize=(14,7))\n\nplt.imshow(wordcloud)\n\nplt.axis('off')\n\nplt.title('Cyberbullying WordCloud')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:11.273307Z","iopub.execute_input":"2026-05-10T19:24:11.273571Z","iopub.status.idle":"2026-05-10T19:24:48.842295Z","shell.execute_reply.started":"2026-05-10T19:24:11.273546Z","shell.execute_reply":"2026-05-10T19:24:48.841368Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## TF-IDF Vectorization","metadata":{}},{"cell_type":"code","source":"vectorizer = TfidfVectorizer(\n\n    max_features=80000,\n\n    ngram_range=(1,2),\n\n    min_df=3,\n\n    max_df=0.9,\n\n    sublinear_tf=True,\n\n    strip_accents='unicode',\n\n    dtype=np.float32\n)\n\nX = vectorizer.fit_transform(\n    combined_df['clean_text']\n)\n\ny = combined_df['label']\n\nprint(X.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:24:48.843325Z","iopub.execute_input":"2026-05-10T19:24:48.843627Z","iopub.status.idle":"2026-05-10T19:25:27.242499Z","shell.execute_reply.started":"2026-05-10T19:24:48.843602Z","shell.execute_reply":"2026-05-10T19:25:27.241700Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train-Test Split","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X,\n    y,\n    test_size=0.2,\n    random_state=42,\n    stratify=y\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:25:27.243500Z","iopub.execute_input":"2026-05-10T19:25:27.243891Z","iopub.status.idle":"2026-05-10T19:25:27.413123Z","shell.execute_reply.started":"2026-05-10T19:25:27.243855Z","shell.execute_reply":"2026-05-10T19:25:27.412387Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handle Class Imbalance","metadata":{}},{"cell_type":"code","source":"counter = Counter(y)\n\nscale_pos_weight = counter[0] / counter[1]\n\nprint(counter)\n\nprint(scale_pos_weight)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:25:27.414299Z","iopub.execute_input":"2026-05-10T19:25:27.415088Z","iopub.status.idle":"2026-05-10T19:25:27.457695Z","shell.execute_reply.started":"2026-05-10T19:25:27.415057Z","shell.execute_reply":"2026-05-10T19:25:27.456819Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Models","metadata":{}},{"cell_type":"code","source":"# Logistic Regression\nlr_model = LogisticRegression(\n    max_iter=1000\n)\n\nlr_model.fit(X_train, y_train)\n\nlr_pred = lr_model.predict(X_test)\n\n# Naive Bayes\nnb_model = MultinomialNB()\n\nnb_model.fit(X_train, y_train)\n\nnb_pred = nb_model.predict(X_test)\n\n# Random Forest\nrf_model = RandomForestClassifier(\n\n    n_estimators=50,\n\n    max_depth=20,\n\n    n_jobs=-1,\n\n    random_state=42\n)\n\nrf_model.fit(X_train, y_train)\n\nrf_pred = rf_model.predict(X_test)\n\n# SVM\nsvm_model = LinearSVC()\n\nsvm_model.fit(X_train, y_train)\n\nsvm_pred = svm_model.predict(X_test)\n\n# GPU XGBoost\nxgb_model = XGBClassifier(\n\n    tree_method='hist',\n    device='cuda',\n\n    n_estimators=1000,\n\n    learning_rate=0.03,\n\n    max_depth=12,\n\n    min_child_weight=2,\n\n    subsample=0.85,\n\n    colsample_bytree=0.85,\n\n    gamma=0.15,\n\n    scale_pos_weight=scale_pos_weight,\n\n    eval_metric='logloss',\n\n    random_state=42,\n\n    n_jobs=-1\n)\n\nxgb_model.fit(\n    X_train,\n    y_train\n)\n\ny_probs = xgb_model.predict_proba(X_test)[:,1]\n\nxgb_pred = (y_probs > 0.35).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:25:27.458841Z","iopub.execute_input":"2026-05-10T19:25:27.459189Z","iopub.status.idle":"2026-05-10T19:27:02.689004Z","shell.execute_reply.started":"2026-05-10T19:25:27.459147Z","shell.execute_reply":"2026-05-10T19:27:02.688080Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Classification Reports","metadata":{}},{"cell_type":"code","source":"print('LOGISTIC REGRESSION')\nprint(classification_report(y_test, lr_pred))\n\nprint('NAIVE BAYES')\nprint(classification_report(y_test, nb_pred))\n\nprint('RANDOM FOREST')\nprint(classification_report(y_test, rf_pred))\n\nprint('SVM')\nprint(classification_report(y_test, svm_pred))\n\nprint('XGBOOST')\nprint(classification_report(y_test, xgb_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:27:02.690129Z","iopub.execute_input":"2026-05-10T19:27:02.690713Z","iopub.status.idle":"2026-05-10T19:27:02.783595Z","shell.execute_reply.started":"2026-05-10T19:27:02.690669Z","shell.execute_reply":"2026-05-10T19:27:02.782697Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Metrics Comparison","metadata":{}},{"cell_type":"code","source":"metrics_df = pd.DataFrame({\n\n    'Model': [\n        'Logistic Regression',\n        'Naive Bayes',\n        'Random Forest',\n        'SVM',\n        'XGBoost'\n    ],\n\n    'Accuracy': [\n        accuracy_score(y_test, lr_pred),\n        accuracy_score(y_test, nb_pred),\n        accuracy_score(y_test, rf_pred),\n        accuracy_score(y_test, svm_pred),\n        accuracy_score(y_test, xgb_pred)\n    ],\n\n    'Precision': [\n        precision_score(y_test, lr_pred),\n        precision_score(y_test, nb_pred),\n        precision_score(y_test, rf_pred),\n        precision_score(y_test, svm_pred),\n        precision_score(y_test, xgb_pred)\n    ],\n\n    'Recall': [\n        recall_score(y_test, lr_pred),\n        recall_score(y_test, nb_pred),\n        recall_score(y_test, rf_pred),\n        recall_score(y_test, svm_pred),\n        recall_score(y_test, xgb_pred)\n    ],\n\n    'F1-Score': [\n        f1_score(y_test, lr_pred),\n        f1_score(y_test, nb_pred),\n        f1_score(y_test, rf_pred),\n        f1_score(y_test, svm_pred),\n        f1_score(y_test, xgb_pred)\n    ]\n})\n\nmetrics_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T19:27:02.784703Z","iopub.execute_input":"2026-05-10T19:27:02.785413Z","iopub.status.idle":"2026-05-10T19:27:02.892004Z","shell.execute_reply.started":"2026-05-10T19:27:02.785384Z","shell.execute_reply":"2026-05-10T19:27:02.891273Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Accuracy Comparison Graph","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10,5))\n\nsns.barplot(\n    x='Model',\n    y='Accuracy',\n    data=metrics_df\n)\n\nplt.title('Accuracy Comparison of ML Models')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Precision Recall F1 Comparison","metadata":{}},{"cell_type":"code","source":"metrics_melted = metrics_df.melt(\n    id_vars='Model',\n    value_vars=['Precision', 'Recall', 'F1-Score']\n)\n\nplt.figure(figsize=(12,6))\n\nsns.barplot(\n    x='Model',\n    y='value',\n    hue='variable',\n    data=metrics_melted\n)\n\nplt.title('Precision Recall and F1 Score Comparison')\n\nplt.ylabel('Score')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"cell_type":"code","source":"cm = confusion_matrix(y_test, xgb_pred)\n\nplt.figure(figsize=(6,5))\n\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt='d',\n    cmap='Blues'\n)\n\nplt.title('Confusion Matrix - XGBoost')\n\nplt.xlabel('Predicted')\n\nplt.ylabel('Actual')\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(\n    y_test,\n    y_probs\n)\n\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7,6))\n\nplt.plot(\n    fpr,\n    tpr,\n    label=f'AUC = {roc_auc:.2f}'\n)\n\nplt.plot([0,1], [0,1], linestyle='--')\n\nplt.xlabel('False Positive Rate')\n\nplt.ylabel('True Positive Rate')\n\nplt.title('ROC Curve - XGBoost')\n\nplt.legend()\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"cell_type":"code","source":"importance = xgb_model.feature_importances_\n\nfeature_names = vectorizer.get_feature_names_out()\n\nfeature_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': importance\n})\n\nfeature_df = feature_df.sort_values(\n    by='Importance',\n    ascending=False\n).head(20)\n\nplt.figure(figsize=(10,8))\n\nsns.barplot(\n    x='Importance',\n    y='Feature',\n    data=feature_df\n)\n\nplt.title('Top 20 Important Features')\n\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Auto Select Best Model","metadata":{}},{"cell_type":"code","source":"models = {\n    'Logistic Regression': lr_model,\n    'Naive Bayes': nb_model,\n    'Random Forest': rf_model,\n    'SVM': svm_model,\n    'XGBoost': xgb_model\n}\n\npredictions = {\n    'Logistic Regression': lr_pred,\n    'Naive Bayes': nb_pred,\n    'Random Forest': rf_pred,\n    'SVM': svm_pred,\n    'XGBoost': xgb_pred\n}\n\nmodel_scores = {}\n\nfor name in models:\n\n    score = f1_score(\n        y_test,\n        predictions[name]\n    )\n\n    model_scores[name] = score\n\nbest_model_name = max(\n    model_scores,\n    key=model_scores.get\n)\n\nbest_model = models[best_model_name]\n\nprint('Best Model:', best_model_name)\n\nprint('Best F1 Score:', model_scores[best_model_name])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Real Time Prediction","metadata":{}},{"cell_type":"code","source":"def predict_cyberbullying(text):\n\n    cleaned = clean_text(text)\n\n    vectorized = vectorizer.transform([cleaned])\n\n    if hasattr(best_model, 'predict_proba'):\n\n        probability = best_model.predict_proba(vectorized)[0][1]\n\n        prediction = 1 if probability > 0.35 else 0\n\n        score = probability\n\n    elif hasattr(best_model, 'decision_function'):\n\n        score = best_model.decision_function(vectorized)[0]\n\n        prediction = 1 if score > -0.5 else 0\n\n    else:\n\n        prediction = best_model.predict(vectorized)[0]\n\n        score = 0\n\n    if prediction == 1:\n\n        return (\n            f'Cyberbullying Detected '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\n    else:\n\n        return (\n            f'Non-Cyberbullying '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\nprint(predict_cyberbullying(\"You are useless and nobody likes you\"))\n\nprint(predict_cyberbullying(\"Go kill yourself\"))\n\nprint(predict_cyberbullying(\"Have a nice day\"))\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}