{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10737,"databundleVersionId":290346,"sourceType":"competition"}],"dockerImageVersionId":30886,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('../input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-15T19:58:35.635161Z","iopub.execute_input":"2025-02-15T19:58:35.635553Z","iopub.status.idle":"2025-02-15T19:58:35.649149Z","shell.execute_reply.started":"2025-02-15T19:58:35.635524Z","shell.execute_reply":"2025-02-15T19:58:35.648013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport re\nimport string\nimport nltk\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom imblearn.over_sampling import SMOTE\nfrom sentence_transformers import SentenceTransformer\n\n# Download NLTK resources\nnltk.download(\"punkt\")\nnltk.download(\"wordnet\")\nnltk.download(\"stopwords\")\n\n# Load dataset\ndf = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T19:58:41.148130Z","iopub.execute_input":"2025-02-15T19:58:41.148520Z","iopub.status.idle":"2025-02-15T19:58:46.501088Z","shell.execute_reply.started":"2025-02-15T19:58:41.148488Z","shell.execute_reply":"2025-02-15T19:58:46.499470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport re\nimport string\nimport nltk\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom imblearn.over_sampling import SMOTE\nfrom sentence_transformers import SentenceTransformer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n\n# Download NLTK resources\nnltk.download(\"punkt\")\nnltk.download(\"wordnet\")\nnltk.download(\"stopwords\")\n\n# Load dataset\ndf = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\").sample(10000)\n\n# Custom text preprocessing\ndef preprocess_text(text):\n    text = text.lower()  # Lowercasing\n    text = re.sub(r'\\d+', '', text)  # Remove numbers\n    text = text.translate(str.maketrans('', '', string.punctuation))  # Remove punctuation\n    tokens = text.split()  # Tokenization\n    stop_words = set(nltk.corpus.stopwords.words(\"english\"))\n    tokens = [word for word in tokens if word not in stop_words]  # Remove stopwords\n    return \" \".join(tokens)\n\n# Apply text preprocessing\ndf[\"clean_text\"] = df[\"question_text\"].apply(preprocess_text)\n\n# TF-IDF Vectorization\ntfidf_vectorizer = TfidfVectorizer(max_features=5000)\nX_tfidf = tfidf_vectorizer.fit_transform(df[\"clean_text\"])\ny = df[\"target\"].values\n\n# BERT Embeddings\nbert_model = SentenceTransformer(\"all-MiniLM-L6-v2\")\nX_bert = bert_model.encode(df[\"clean_text\"].tolist(), show_progress_bar=True)\n\n# Standardize BERT features\nscaler = StandardScaler()\nX_bert_scaled = scaler.fit_transform(X_bert)\n\n# Apply PCA to reduce dimensionality\npca = PCA(n_components=100)\nX_bert_pca = pca.fit_transform(X_bert_scaled)\n\n# Split data\nX_train_tfidf, X_test_tfidf, y_train, y_test = train_test_split(X_tfidf, y, test_size=0.2, random_state=42)\nX_train_bert, X_test_bert, _, _ = train_test_split(X_bert_pca, y, test_size=0.2, random_state=42)\n\n# Handle class imbalance using SMOTE\nsmote = SMOTE(random_state=42)\nX_train_tfidf_resampled, y_train_resampled = smote.fit_resample(X_train_tfidf, y_train)\nX_train_bert_resampled, y_train_bert_resampled = smote.fit_resample(X_train_bert, y_train)\n\n# Train Logistic Regression on TF-IDF\nlog_reg = LogisticRegression()\nlog_reg.fit(X_train_tfidf_resampled, y_train_resampled)\n\ny_pred_tfidf = log_reg.predict(X_test_tfidf)\nprint(\"Logistic Regression (TF-IDF)\\n\", classification_report(y_test, y_pred_tfidf))\n\n# Train Gradient Boosting Classifier on BERT\ngbc = GradientBoostingClassifier(n_estimators=200, learning_rate=0.1, random_state=42)\ngbc.fit(X_train_bert_resampled, y_train_bert_resampled)\n\ny_pred_bert = gbc.predict(X_test_bert)\nprint(\"Gradient Boosting (BERT)\\n\", classification_report(y_test, y_pred_bert))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T19:59:16.978604Z","iopub.execute_input":"2025-02-15T19:59:16.979052Z","iopub.status.idle":"2025-02-15T20:02:18.651415Z","shell.execute_reply.started":"2025-02-15T19:59:16.979022Z","shell.execute_reply":"2025-02-15T20:02:18.650194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport re\nimport string\nimport nltk\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom imblearn.over_sampling import SMOTE\nfrom sentence_transformers import SentenceTransformer\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\n\n\n# Standardize BERT features\nX_bert_scaled = scaler.fit_transform(X_bert)\n\n# Apply PCA to reduce dimensionality\npca = PCA(n_components=100)\nX_bert_pca = pca.fit_transform(X_bert_scaled)\n\n# Split data\nX_train_bert, X_test_bert, _, _ = train_test_split(X_bert_pca, y, test_size=0.2, random_state=42)\n\n# Handle class imbalance using SMOTE\nsmote = SMOTE(random_state=42)\nX_train_bert_resampled, y_train_bert_resampled = smote.fit_resample(X_train_bert, y_train)\n\n# Train Gradient Boosting Classifier on BERT\ngbc = GradientBoostingClassifier(n_estimators=200, learning_rate=0.1, random_state=42)\ngbc.fit(X_train_bert_resampled, y_train_bert_resampled)\n\ny_pred_bert = gbc.predict(X_test_bert)\nprint(\"Gradient Boosting (BERT)\\n\", classification_report(y_test, y_pred_bert))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-15T20:02:18.653124Z","iopub.execute_input":"2025-02-15T20:02:18.653552Z","iopub.status.idle":"2025-02-15T20:04:32.814486Z","shell.execute_reply.started":"2025-02-15T20:02:18.653512Z","shell.execute_reply":"2025-02-15T20:04:32.813341Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}