{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":19018,"databundleVersionId":2703900},{"sourceType":"datasetVersion","sourceId":3053020,"datasetId":1869236,"databundleVersionId":3101552},{"sourceType":"datasetVersion","sourceId":1257215,"datasetId":723100,"databundleVersionId":1289103},{"sourceType":"datasetVersion","sourceId":2798066,"datasetId":1709138,"databundleVersionId":2844101}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"69cb3604","cell_type":"markdown","source":"# Advanced Cyberbullying Detection Using Machine Learning\n\nComplete project notebook with:\n- Multi-dataset training\n- Multilingual toxic comments\n- TF-IDF vectorization\n- Multiple ML models\n- GPU XGBoost\n- Automatic best model selection\n- Advanced evaluation graphs\n- Real-time prediction\n","metadata":{}},{"id":"6d86bce6","cell_type":"code","source":"!pip install nltk xgboost scikit-learn pandas numpy matplotlib seaborn wordcloud imbalanced-learn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:42.254938Z","iopub.execute_input":"2026-05-10T06:00:42.255790Z","iopub.status.idle":"2026-05-10T06:00:46.196114Z","shell.execute_reply.started":"2026-05-10T06:00:42.255752Z","shell.execute_reply":"2026-05-10T06:00:46.195313Z"}},"outputs":[],"execution_count":null},{"id":"6d713c4b","cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport nltk\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom collections import Counter\n\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.metrics import (\n    accuracy_score,\n    precision_score,\n    recall_score,\n    f1_score,\n    confusion_matrix,\n    classification_report,\n    roc_curve,\n    auc\n)\n\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.svm import LinearSVC\n\nfrom xgboost import XGBClassifier\n\nfrom wordcloud import WordCloud\n\nnltk.download('stopwords')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:46.198135Z","iopub.execute_input":"2026-05-10T06:00:46.198565Z","iopub.status.idle":"2026-05-10T06:00:46.208156Z","shell.execute_reply.started":"2026-05-10T06:00:46.198532Z","shell.execute_reply":"2026-05-10T06:00:46.207318Z"}},"outputs":[],"execution_count":null},{"id":"912e0fef","cell_type":"markdown","source":"## Load Datasets","metadata":{}},{"id":"d7fc5151","cell_type":"code","source":"path1 = '/kaggle/input/datasets/andrewmvd/cyberbullying-classification/cyberbullying_tweets.csv'\n\npath2 = '/kaggle/input/datasets/mrmorj/hate-speech-and-offensive-language-dataset/labeled_data.csv'\n\npath3 = '/kaggle/input/datasets/julian3833/jigsaw-toxic-comment-classification-challenge/train.csv'\n\npath4 = '/kaggle/input/competitions/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv'\n\ndf1 = pd.read_csv(path1)\ndf2 = pd.read_csv(path2)\ndf3 = pd.read_csv(path3)\ndf4 = pd.read_csv(path4)\n\nprint(df1.shape)\nprint(df2.shape)\nprint(df3.shape)\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:46.209348Z","iopub.execute_input":"2026-05-10T06:00:46.209569Z","iopub.status.idle":"2026-05-10T06:00:49.348896Z","shell.execute_reply.started":"2026-05-10T06:00:46.209548Z","shell.execute_reply":"2026-05-10T06:00:49.348157Z"}},"outputs":[],"execution_count":null},{"id":"1831efda","cell_type":"markdown","source":"## Dataset Preprocessing","metadata":{}},{"id":"447ecf8a","cell_type":"code","source":"# DATASET 1\ndf1 = df1.rename(columns={\n    'tweet_text':'text',\n    'cyberbullying_type':'label'\n})\n\ndf1 = df1[['text', 'label']]\n\ndf1['label'] = df1['label'].apply(\n    lambda x: 0 if x == 'not_cyberbullying' else 1\n)\n\n# DATASET 2\ndf2 = df2.rename(columns={\n    'tweet':'text',\n    'class':'label'\n})\n\ndf2 = df2[['text', 'label']]\n\ndf2['label'] = df2['label'].apply(\n    lambda x: 0 if x == 2 else 1\n)\n\n# DATASET 3\ndf3['label'] = (\n    df3[['toxic', 'severe_toxic',\n         'obscene', 'threat',\n         'insult', 'identity_hate']]\n    .sum(axis=1)\n)\n\ndf3['label'] = df3['label'].apply(\n    lambda x: 1 if x > 0 else 0\n)\n\ndf3 = df3.rename(columns={\n    'comment_text':'text'\n})\n\ndf3 = df3[['text', 'label']]\n\n# DATASET 4\ndf4 = df4.rename(columns={\n    'comment_text':'text',\n    'toxic':'label'\n})\n\ndf4 = df4[['text', 'label']]\n\ndf4.dropna(inplace=True)\n\n# Sample multilingual dataset for RAM optimization\ndf4 = df4.sample(\n    n=100000,\n    random_state=42\n)\n\nprint(df4.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:49.349929Z","iopub.execute_input":"2026-05-10T06:00:49.350259Z","iopub.status.idle":"2026-05-10T06:00:49.613339Z","shell.execute_reply.started":"2026-05-10T06:00:49.350220Z","shell.execute_reply":"2026-05-10T06:00:49.612037Z"}},"outputs":[],"execution_count":null},{"id":"de4bf472","cell_type":"markdown","source":"## Merge Datasets","metadata":{}},{"id":"34aba61e","cell_type":"code","source":"combined_df = pd.concat(\n    [df1, df2, df3, df4],\n    ignore_index=True\n)\n\ncombined_df.drop_duplicates(inplace=True)\ncombined_df.dropna(inplace=True)\n\ncombined_df['label'] = combined_df['label'].astype(int)\n\ncombined_df = combined_df[\n    combined_df['label'].isin([0,1])\n]\n\nprint(combined_df.shape)\n\nprint(combined_df['label'].value_counts())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:49.617153Z","iopub.execute_input":"2026-05-10T06:00:49.617711Z","iopub.status.idle":"2026-05-10T06:00:50.190630Z","shell.execute_reply.started":"2026-05-10T06:00:49.617653Z","shell.execute_reply":"2026-05-10T06:00:50.189899Z"}},"outputs":[],"execution_count":null},{"id":"469ffdd3","cell_type":"markdown","source":"## Label Distribution","metadata":{}},{"id":"c00dbbdc","cell_type":"code","source":"plt.figure(figsize=(6,5))\n\nsns.countplot(\n    x=combined_df['label']\n)\n\nplt.title('Cyberbullying vs Non-Cyberbullying')\n\nplt.xticks(\n    [0,1],\n    ['Non-Cyberbullying', 'Cyberbullying']\n)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:50.191743Z","iopub.execute_input":"2026-05-10T06:00:50.192153Z","iopub.status.idle":"2026-05-10T06:00:50.728972Z","shell.execute_reply.started":"2026-05-10T06:00:50.192124Z","shell.execute_reply":"2026-05-10T06:00:50.728101Z"}},"outputs":[],"execution_count":null},{"id":"f2ab24e0","cell_type":"markdown","source":"## Text Cleaning","metadata":{}},{"id":"224feb2e","cell_type":"code","source":"def clean_text(text):\n\n    text = str(text).lower()\n\n    text = re.sub(r'http\\S+', '', text)\n\n    text = re.sub(r'\\n', ' ', text)\n\n    text = re.sub(r'\\s+', ' ', text)\n\n    return text\n\ncombined_df['clean_text'] = combined_df['text'].apply(clean_text)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:50.730145Z","iopub.execute_input":"2026-05-10T06:00:50.730513Z","iopub.status.idle":"2026-05-10T06:00:56.728696Z","shell.execute_reply.started":"2026-05-10T06:00:50.730487Z","shell.execute_reply":"2026-05-10T06:00:56.727861Z"}},"outputs":[],"execution_count":null},{"id":"76fc062a","cell_type":"markdown","source":"## WordCloud","metadata":{}},{"id":"dc5364f9","cell_type":"code","source":"text_data = ' '.join(combined_df['clean_text'])\n\nwordcloud = WordCloud(\n    width=1200,\n    height=600,\n    background_color='white'\n).generate(text_data)\n\nplt.figure(figsize=(14,7))\n\nplt.imshow(wordcloud)\n\nplt.axis('off')\n\nplt.title('Cyberbullying WordCloud')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:00:56.729661Z","iopub.execute_input":"2026-05-10T06:00:56.729954Z","iopub.status.idle":"2026-05-10T06:01:37.013116Z","shell.execute_reply.started":"2026-05-10T06:00:56.729927Z","shell.execute_reply":"2026-05-10T06:01:37.012104Z"}},"outputs":[],"execution_count":null},{"id":"966861df","cell_type":"markdown","source":"## TF-IDF Vectorization","metadata":{}},{"id":"ecdfe395","cell_type":"code","source":"vectorizer = TfidfVectorizer(\n\n    max_features=80000,\n\n    ngram_range=(1,2),\n\n    min_df=3,\n\n    max_df=0.9,\n\n    sublinear_tf=True,\n\n    strip_accents='unicode',\n\n    dtype=np.float32\n)\n\nX = vectorizer.fit_transform(\n    combined_df['clean_text']\n)\n\ny = combined_df['label']\n\nprint(X.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:01:37.014571Z","iopub.execute_input":"2026-05-10T06:01:37.014890Z","iopub.status.idle":"2026-05-10T06:02:19.610815Z","shell.execute_reply.started":"2026-05-10T06:01:37.014863Z","shell.execute_reply":"2026-05-10T06:02:19.609932Z"}},"outputs":[],"execution_count":null},{"id":"7df3cacf","cell_type":"markdown","source":"## Train-Test Split","metadata":{}},{"id":"b615aafb","cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X,\n    y,\n    test_size=0.2,\n    random_state=42,\n    stratify=y\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:02:19.611917Z","iopub.execute_input":"2026-05-10T06:02:19.612230Z","iopub.status.idle":"2026-05-10T06:02:19.795086Z","shell.execute_reply.started":"2026-05-10T06:02:19.612203Z","shell.execute_reply":"2026-05-10T06:02:19.794363Z"}},"outputs":[],"execution_count":null},{"id":"810dbeb5","cell_type":"markdown","source":"## Handle Class Imbalance","metadata":{}},{"id":"01f17e57","cell_type":"code","source":"counter = Counter(y)\n\nscale_pos_weight = counter[0] / counter[1]\n\nprint(counter)\n\nprint(scale_pos_weight)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:02:19.796199Z","iopub.execute_input":"2026-05-10T06:02:19.796539Z","iopub.status.idle":"2026-05-10T06:02:19.836836Z","shell.execute_reply.started":"2026-05-10T06:02:19.796514Z","shell.execute_reply":"2026-05-10T06:02:19.835892Z"}},"outputs":[],"execution_count":null},{"id":"afd86c5e","cell_type":"markdown","source":"## Train Models","metadata":{}},{"id":"1e064b59","cell_type":"code","source":"# Logistic Regression\nlr_model = LogisticRegression(\n    max_iter=1000\n)\n\nlr_model.fit(X_train, y_train)\n\nlr_pred = lr_model.predict(X_test)\n\n# Naive Bayes\nnb_model = MultinomialNB()\n\nnb_model.fit(X_train, y_train)\n\nnb_pred = nb_model.predict(X_test)\n\n# Random Forest\nrf_model = RandomForestClassifier(\n\n    n_estimators=50,\n\n    max_depth=20,\n\n    n_jobs=-1,\n\n    random_state=42\n)\n\nrf_model.fit(X_train, y_train)\n\nrf_pred = rf_model.predict(X_test)\n\n# SVM\nsvm_model = LinearSVC()\n\nsvm_model.fit(X_train, y_train)\n\nsvm_pred = svm_model.predict(X_test)\n\n# GPU XGBoost\nxgb_model = XGBClassifier(\n\n    tree_method='hist',\n    device='cuda',\n\n    n_estimators=1000,\n\n    learning_rate=0.03,\n\n    max_depth=12,\n\n    min_child_weight=2,\n\n    subsample=0.85,\n\n    colsample_bytree=0.85,\n\n    gamma=0.15,\n\n    scale_pos_weight=scale_pos_weight,\n\n    eval_metric='logloss',\n\n    random_state=42,\n\n    n_jobs=-1\n)\n\nxgb_model.fit(\n    X_train,\n    y_train\n)\n\ny_probs = xgb_model.predict_proba(X_test)[:,1]\n\nxgb_pred = (y_probs > 0.35).astype(int)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:02:19.838074Z","iopub.execute_input":"2026-05-10T06:02:19.838449Z","iopub.status.idle":"2026-05-10T06:03:55.076090Z","shell.execute_reply.started":"2026-05-10T06:02:19.838423Z","shell.execute_reply":"2026-05-10T06:03:55.075395Z"}},"outputs":[],"execution_count":null},{"id":"6ee2d9af","cell_type":"markdown","source":"## Classification Reports","metadata":{}},{"id":"7c5742ab","cell_type":"code","source":"print('LOGISTIC REGRESSION')\nprint(classification_report(y_test, lr_pred))\n\nprint('NAIVE BAYES')\nprint(classification_report(y_test, nb_pred))\n\nprint('RANDOM FOREST')\nprint(classification_report(y_test, rf_pred))\n\nprint('SVM')\nprint(classification_report(y_test, svm_pred))\n\nprint('XGBOOST')\nprint(classification_report(y_test, xgb_pred))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.077103Z","iopub.execute_input":"2026-05-10T06:03:55.077427Z","iopub.status.idle":"2026-05-10T06:03:55.175570Z","shell.execute_reply.started":"2026-05-10T06:03:55.077399Z","shell.execute_reply":"2026-05-10T06:03:55.174793Z"}},"outputs":[],"execution_count":null},{"id":"06e5762a","cell_type":"markdown","source":"## Metrics Comparison","metadata":{}},{"id":"c94bd6e8","cell_type":"code","source":"metrics_df = pd.DataFrame({\n\n    'Model': [\n        'Logistic Regression',\n        'Naive Bayes',\n        'Random Forest',\n        'SVM',\n        'XGBoost'\n    ],\n\n    'Accuracy': [\n        accuracy_score(y_test, lr_pred),\n        accuracy_score(y_test, nb_pred),\n        accuracy_score(y_test, rf_pred),\n        accuracy_score(y_test, svm_pred),\n        accuracy_score(y_test, xgb_pred)\n    ],\n\n    'Precision': [\n        precision_score(y_test, lr_pred),\n        precision_score(y_test, nb_pred),\n        precision_score(y_test, rf_pred),\n        precision_score(y_test, svm_pred),\n        precision_score(y_test, xgb_pred)\n    ],\n\n    'Recall': [\n        recall_score(y_test, lr_pred),\n        recall_score(y_test, nb_pred),\n        recall_score(y_test, rf_pred),\n        recall_score(y_test, svm_pred),\n        recall_score(y_test, xgb_pred)\n    ],\n\n    'F1-Score': [\n        f1_score(y_test, lr_pred),\n        f1_score(y_test, nb_pred),\n        f1_score(y_test, rf_pred),\n        f1_score(y_test, svm_pred),\n        f1_score(y_test, xgb_pred)\n    ]\n})\n\nmetrics_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.178304Z","iopub.execute_input":"2026-05-10T06:03:55.178670Z","iopub.status.idle":"2026-05-10T06:03:55.289179Z","shell.execute_reply.started":"2026-05-10T06:03:55.178642Z","shell.execute_reply":"2026-05-10T06:03:55.288366Z"}},"outputs":[],"execution_count":null},{"id":"8994a9bc","cell_type":"markdown","source":"## Accuracy Comparison Graph","metadata":{}},{"id":"3cb28bfd","cell_type":"code","source":"plt.figure(figsize=(10,5))\n\nsns.barplot(\n    x='Model',\n    y='Accuracy',\n    data=metrics_df\n)\n\nplt.title('Accuracy Comparison of ML Models')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.290202Z","iopub.execute_input":"2026-05-10T06:03:55.290661Z","iopub.status.idle":"2026-05-10T06:03:55.458131Z","shell.execute_reply.started":"2026-05-10T06:03:55.290633Z","shell.execute_reply":"2026-05-10T06:03:55.457219Z"}},"outputs":[],"execution_count":null},{"id":"1de202ca","cell_type":"markdown","source":"## Precision Recall F1 Comparison","metadata":{}},{"id":"a30bf83c","cell_type":"code","source":"metrics_melted = metrics_df.melt(\n    id_vars='Model',\n    value_vars=['Precision', 'Recall', 'F1-Score']\n)\n\nplt.figure(figsize=(12,6))\n\nsns.barplot(\n    x='Model',\n    y='value',\n    hue='variable',\n    data=metrics_melted\n)\n\nplt.title('Precision Recall and F1 Score Comparison')\n\nplt.ylabel('Score')\n\nplt.xticks(rotation=15)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.459287Z","iopub.execute_input":"2026-05-10T06:03:55.459619Z","iopub.status.idle":"2026-05-10T06:03:55.701515Z","shell.execute_reply.started":"2026-05-10T06:03:55.459583Z","shell.execute_reply":"2026-05-10T06:03:55.700789Z"}},"outputs":[],"execution_count":null},{"id":"7f392d98","cell_type":"markdown","source":"## Confusion Matrix","metadata":{}},{"id":"bfec0712","cell_type":"code","source":"cm = confusion_matrix(y_test, xgb_pred)\n\nplt.figure(figsize=(6,5))\n\nsns.heatmap(\n    cm,\n    annot=True,\n    fmt='d',\n    cmap='Blues'\n)\n\nplt.title('Confusion Matrix - XGBoost')\n\nplt.xlabel('Predicted')\n\nplt.ylabel('Actual')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.702751Z","iopub.execute_input":"2026-05-10T06:03:55.703207Z","iopub.status.idle":"2026-05-10T06:03:55.848901Z","shell.execute_reply.started":"2026-05-10T06:03:55.703174Z","shell.execute_reply":"2026-05-10T06:03:55.848120Z"}},"outputs":[],"execution_count":null},{"id":"3aab0e88","cell_type":"markdown","source":"## ROC Curve","metadata":{}},{"id":"808f6ff4","cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(\n    y_test,\n    y_probs\n)\n\nroc_auc = auc(fpr, tpr)\n\nplt.figure(figsize=(7,6))\n\nplt.plot(\n    fpr,\n    tpr,\n    label=f'AUC = {roc_auc:.2f}'\n)\n\nplt.plot([0,1], [0,1], linestyle='--')\n\nplt.xlabel('False Positive Rate')\n\nplt.ylabel('True Positive Rate')\n\nplt.title('ROC Curve - XGBoost')\n\nplt.legend()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:55.850638Z","iopub.execute_input":"2026-05-10T06:03:55.850909Z","iopub.status.idle":"2026-05-10T06:03:56.002794Z","shell.execute_reply.started":"2026-05-10T06:03:55.850882Z","shell.execute_reply":"2026-05-10T06:03:56.001805Z"}},"outputs":[],"execution_count":null},{"id":"1eaafe5c","cell_type":"markdown","source":"## Feature Importance","metadata":{}},{"id":"38094eca","cell_type":"code","source":"importance = xgb_model.feature_importances_\n\nfeature_names = vectorizer.get_feature_names_out()\n\nfeature_df = pd.DataFrame({\n    'Feature': feature_names,\n    'Importance': importance\n})\n\nfeature_df = feature_df.sort_values(\n    by='Importance',\n    ascending=False\n).head(20)\n\nplt.figure(figsize=(10,8))\n\nsns.barplot(\n    x='Importance',\n    y='Feature',\n    data=feature_df\n)\n\nplt.title('Top 20 Important Features')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:56.004209Z","iopub.execute_input":"2026-05-10T06:03:56.004594Z","iopub.status.idle":"2026-05-10T06:03:56.414519Z","shell.execute_reply.started":"2026-05-10T06:03:56.004566Z","shell.execute_reply":"2026-05-10T06:03:56.413768Z"}},"outputs":[],"execution_count":null},{"id":"6c7442e9","cell_type":"markdown","source":"## Auto Select Best Model","metadata":{}},{"id":"a203f17e","cell_type":"code","source":"models = {\n    'Logistic Regression': lr_model,\n    'Naive Bayes': nb_model,\n    'Random Forest': rf_model,\n    'SVM': svm_model,\n    'XGBoost': xgb_model\n}\n\npredictions = {\n    'Logistic Regression': lr_pred,\n    'Naive Bayes': nb_pred,\n    'Random Forest': rf_pred,\n    'SVM': svm_pred,\n    'XGBoost': xgb_pred\n}\n\nmodel_scores = {}\n\nfor name in models:\n\n    score = f1_score(\n        y_test,\n        predictions[name]\n    )\n\n    model_scores[name] = score\n\nbest_model_name = max(\n    model_scores,\n    key=model_scores.get\n)\n\nbest_model = models[best_model_name]\n\nprint('Best Model:', best_model_name)\n\nprint('Best F1 Score:', model_scores[best_model_name])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:03:56.415541Z","iopub.execute_input":"2026-05-10T06:03:56.415934Z","iopub.status.idle":"2026-05-10T06:03:56.455592Z","shell.execute_reply.started":"2026-05-10T06:03:56.415892Z","shell.execute_reply":"2026-05-10T06:03:56.454918Z"}},"outputs":[],"execution_count":null},{"id":"1c297626","cell_type":"markdown","source":"## Real Time Prediction","metadata":{}},{"id":"15e459cf","cell_type":"code","source":"def predict_cyberbullying(text):\n\n    cleaned = clean_text(text)\n\n    vectorized = vectorizer.transform([cleaned])\n\n    if hasattr(best_model, 'predict_proba'):\n\n        probability = best_model.predict_proba(vectorized)[0][1]\n\n        prediction = 1 if probability > 0.35 else 0\n\n        score = probability\n\n    elif hasattr(best_model, 'decision_function'):\n\n        score = best_model.decision_function(vectorized)[0]\n\n        prediction = 1 if score > -0.5 else 0\n\n    else:\n\n        prediction = best_model.predict(vectorized)[0]\n\n        score = 0\n\n    if prediction == 1:\n\n        return (\n            f'Cyberbullying Detected '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\n    else:\n\n        return (\n            f'Non-Cyberbullying '\n            f'using {best_model_name} '\n            f'(Score: {score:.2f})'\n        )\n\nprint(predict_cyberbullying(\"You are useless and nobody likes you\"))\n\nprint(predict_cyberbullying(\"Go kill yourself\"))\n\nprint(predict_cyberbullying(\"Have a nice day\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-10T06:05:39.328403Z","iopub.execute_input":"2026-05-10T06:05:39.329337Z","iopub.status.idle":"2026-05-10T06:05:39.340211Z","shell.execute_reply.started":"2026-05-10T06:05:39.329303Z","shell.execute_reply":"2026-05-10T06:05:39.339418Z"}},"outputs":[],"execution_count":null},{"id":"7e3e0bb0-e23d-4f67-b4f2-c036d1d7214b","cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}