{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nfrom string import punctuation\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.model_selection import KFold, StratifiedKFold, train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score,roc_auc_score\nfrom xgboost import XGBClassifier\nimport optuna","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-11T16:04:33.538478Z","iopub.execute_input":"2022-07-11T16:04:33.538906Z","iopub.status.idle":"2022-07-11T16:04:34.767011Z","shell.execute_reply.started":"2022-07-11T16:04:33.538819Z","shell.execute_reply":"2022-07-11T16:04:34.766069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imported some NLTK libraries for data preprocessing and cleaning\nfrom nltk.corpus import stopwords\nfrom nltk.stem.porter import PorterStemmer\nfrom nltk.stem import WordNetLemmatizer\n\nStemmer = PorterStemmer()\nlemmatizer = WordNetLemmatizer()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:34.769073Z","iopub.execute_input":"2022-07-11T16:04:34.769701Z","iopub.status.idle":"2022-07-11T16:04:35.147609Z","shell.execute_reply.started":"2022-07-11T16:04:34.769664Z","shell.execute_reply":"2022-07-11T16:04:35.146509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/nlp-getting-started/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:35.149427Z","iopub.execute_input":"2022-07-11T16:04:35.149726Z","iopub.status.idle":"2022-07-11T16:04:35.194352Z","shell.execute_reply.started":"2022-07-11T16:04:35.149701Z","shell.execute_reply":"2022-07-11T16:04:35.193480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:35.728960Z","iopub.execute_input":"2022-07-11T16:04:35.729308Z","iopub.status.idle":"2022-07-11T16:04:35.749900Z","shell.execute_reply.started":"2022-07-11T16:04:35.729278Z","shell.execute_reply":"2022-07-11T16:04:35.749010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:36.507573Z","iopub.execute_input":"2022-07-11T16:04:36.507929Z","iopub.status.idle":"2022-07-11T16:04:36.532284Z","shell.execute_reply.started":"2022-07-11T16:04:36.507900Z","shell.execute_reply":"2022-07-11T16:04:36.531379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:37.663136Z","iopub.execute_input":"2022-07-11T16:04:37.665077Z","iopub.status.idle":"2022-07-11T16:04:37.677236Z","shell.execute_reply.started":"2022-07-11T16:04:37.665031Z","shell.execute_reply":"2022-07-11T16:04:37.675908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:38.116311Z","iopub.execute_input":"2022-07-11T16:04:38.116876Z","iopub.status.idle":"2022-07-11T16:04:38.124690Z","shell.execute_reply.started":"2022-07-11T16:04:38.116843Z","shell.execute_reply":"2022-07-11T16:04:38.123425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info(show_counts=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:39.235697Z","iopub.execute_input":"2022-07-11T16:04:39.236905Z","iopub.status.idle":"2022-07-11T16:04:39.263350Z","shell.execute_reply.started":"2022-07-11T16:04:39.236858Z","shell.execute_reply":"2022-07-11T16:04:39.262175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"text\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:40.782278Z","iopub.execute_input":"2022-07-11T16:04:40.783199Z","iopub.status.idle":"2022-07-11T16:04:40.792205Z","shell.execute_reply.started":"2022-07-11T16:04:40.783163Z","shell.execute_reply":"2022-07-11T16:04:40.791043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop([\"keyword\", \"location\"], axis=1, inplace=True)\ntest_data.drop([\"keyword\", \"location\"], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:42.224048Z","iopub.execute_input":"2022-07-11T16:04:42.224516Z","iopub.status.idle":"2022-07-11T16:04:42.235100Z","shell.execute_reply.started":"2022-07-11T16:04:42.224477Z","shell.execute_reply":"2022-07-11T16:04:42.233798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### I am just modelling on text column as remaining 2 columns has many missing values","metadata":{}},{"cell_type":"markdown","source":"# Preprocess the data by removing the unnecessary symbols, urls","metadata":{}},{"cell_type":"code","source":"def preprocess_text(text):\n    \n    text = text.lower()\n    text = re.sub('[^A-Za-z0-9]', ' ',text)\n    text = re.sub('[?|$|.|!@#^]',' ',text)\n    text = re.sub('http://\\S+|https://\\S+', ' ',text)\n    text = text.split(\" \")\n    text = [lemmatizer.lemmatize(word) for word in text if word not in stopwords.words('english')\n            and word not in punctuation]\n    text = \" \".join(text)\n    \n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:47.727805Z","iopub.execute_input":"2022-07-11T16:04:47.728146Z","iopub.status.idle":"2022-07-11T16:04:47.734781Z","shell.execute_reply.started":"2022-07-11T16:04:47.728114Z","shell.execute_reply":"2022-07-11T16:04:47.733832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning the text data from the training dataset\ntrain_data[\"text\"] = train_data[\"text\"].apply(preprocess_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:04:48.601954Z","iopub.execute_input":"2022-07-11T16:04:48.602660Z","iopub.status.idle":"2022-07-11T16:05:12.013745Z","shell.execute_reply.started":"2022-07-11T16:04:48.602618Z","shell.execute_reply":"2022-07-11T16:05:12.012506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning the text data from the test dataset\ntest_data[\"text\"] = test_data[\"text\"].apply(preprocess_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:05:12.016084Z","iopub.execute_input":"2022-07-11T16:05:12.016792Z","iopub.status.idle":"2022-07-11T16:05:25.031423Z","shell.execute_reply.started":"2022-07-11T16:05:12.016736Z","shell.execute_reply":"2022-07-11T16:05:25.030414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kfold = KFold(n_splits=5,shuffle=True,random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:05:25.032627Z","iopub.execute_input":"2022-07-11T16:05:25.033004Z","iopub.status.idle":"2022-07-11T16:05:25.039858Z","shell.execute_reply.started":"2022-07-11T16:05:25.032968Z","shell.execute_reply":"2022-07-11T16:05:25.037730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model-1 : XGBClassifier with CountVectorizer","metadata":{}},{"cell_type":"code","source":"final_predictions = []\nf1Score = []\n\nfor fold,(train_idx,valid_idx) in enumerate(kfold.split(train_data, train_data.target)):\n    \n    X_train = train_data.iloc[train_idx][\"text\"]\n    X_valid = train_data.iloc[valid_idx][\"text\"]\n    y_train = train_data.iloc[train_idx][\"target\"]\n    y_valid = train_data.iloc[valid_idx][\"target\"]\n    \n#     X_test = test_data[\"text\"].copy()\n#     X_test = X_test.drop(\"id\",axis=1)\n    \n    vectorizer = CountVectorizer(max_features=1000)\n    \n    trainVectors = vectorizer.fit_transform(X_train).toarray()\n    validVectors = vectorizer.transform(X_valid).toarray()\n#     testVectors  = vectorizer.transform(X_test).toarray()\n    \n    y_train_vector = np.array(y_train)\n    y_valid_vector = np.array(y_valid)\n    \n    model = XGBClassifier(\n                        random_state=fold,\n                        objective='binary:logistic',\n                        tree_method='gpu_hist',  \n                        gpu_id=0,\n                        predictor='gpu_predictor',\n                        n_jobs = -1\n                        )\n    \n    model.fit(trainVectors,y_train_vector)\n    \n    predictions = model.predict(validVectors)\n    \n    print(f\"F1 Score after {fold} fold is {f1_score(y_valid_vector, predictions)}\")\n    \n    f1Score.append(f1_score(y_valid_vector, predictions))\n    \n#     testPredictions = model.predict(testVectors)\n    \n#     final_predictions.append(testPredictions)\n    \nprint(f\"The mean F1 Score is {np.mean(f1Score)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T07:55:51.678185Z","iopub.execute_input":"2022-07-11T07:55:51.678668Z","iopub.status.idle":"2022-07-11T07:55:57.150251Z","shell.execute_reply.started":"2022-07-11T07:55:51.678601Z","shell.execute_reply":"2022-07-11T07:55:57.148994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model-4 : XGBClassifier with TFIDFVectorizer","metadata":{}},{"cell_type":"code","source":"final_predictions = []\nf1Score = []\n\nfor fold,(train_idx,valid_idx) in enumerate(kfold.split(train_data, train_data.target)):\n    \n    X_train = train_data.iloc[train_idx][\"text\"]\n    X_valid = train_data.iloc[valid_idx][\"text\"]\n    y_train = train_data.iloc[train_idx][\"target\"]\n    y_valid = train_data.iloc[valid_idx][\"target\"]\n    \n    vectorizer = TfidfVectorizer(max_features=1000)\n    \n    trainVectors = vectorizer.fit_transform(X_train).toarray()\n    validVectors = vectorizer.transform(X_valid).toarray()\n    \n    y_train_vector = np.array(y_train)\n    y_valid_vector = np.array(y_valid)\n    \n    model = XGBClassifier(\n                        random_state=fold,\n                        objective='binary:logistic',\n                        tree_method='gpu_hist',  \n                        gpu_id=0,\n                        predictor='gpu_predictor',\n                        n_jobs = -1\n                        )\n    \n    model.fit(trainVectors,y_train_vector)\n    \n    predictions = model.predict(validVectors)\n    \n    print(f\"F1 Score after {fold} fold is {f1_score(y_valid_vector, predictions)}\")\n    \n    f1Score.append(f1_score(y_valid_vector, predictions))\n    \nprint(f\"The mean F1 Score is {np.mean(f1Score)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T07:56:34.205477Z","iopub.execute_input":"2022-07-11T07:56:34.206316Z","iopub.status.idle":"2022-07-11T07:56:41.039510Z","shell.execute_reply.started":"2022-07-11T07:56:34.206262Z","shell.execute_reply":"2022-07-11T07:56:41.038310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Finding best params for XGB using optuna","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n\n    final_predictions = []\n    f1Score = []\n\n    for fold,(train_idx,valid_idx) in enumerate(kfold.split(train_data, train_data.target)):\n        \n        params = {\n            \"learning_rate\" : trial.suggest_float(\"learning_rate\", 1e-4, 0.5, log=True),\n            \"reg_lambda\" : trial.suggest_loguniform(\"reg_lambda\", 1e-8, 50.0),\n            \"reg_alpha\" : trial.suggest_loguniform(\"reg_alpha\", 1e-8, 50.0),\n            \"subsample\" : trial.suggest_float(\"subsample\", 0.1, 1.0),\n            \"colsample_bytree\" : trial.suggest_float(\"colsample_bytree\", 0.1, 1.0),\n            \"max_depth\" : trial.suggest_int(\"max_depth\", 1, 10),\n            \"n_estimators\" : trial.suggest_int(\"n_estimators\",1000,20000)\n            }\n        X_train = train_data.iloc[train_idx][\"text\"]\n        X_valid = train_data.iloc[valid_idx][\"text\"]\n        y_train = train_data.iloc[train_idx][\"target\"]\n        y_valid = train_data.iloc[valid_idx][\"target\"]\n\n        vectorizer = CountVectorizer(max_features=1000)\n\n        trainVectors = vectorizer.fit_transform(X_train).toarray()\n        validVectors = vectorizer.transform(X_valid).toarray()\n        testVectors  = vectorizer.transform(X_test).toarray()\n\n        y_train_vector = np.array(y_train)\n        y_valid_vector = np.array(y_valid)\n\n        model = XGBClassifier(\n                            random_state=fold,\n                            objective='binary:logistic',\n                            tree_method='gpu_hist',  \n                            gpu_id=0,\n                            predictor='gpu_predictor',\n                            n_jobs = -1,\n                            **params\n                            )\n\n        model.fit(trainVectors,y_train_vector)\n\n        predictions = model.predict(validVectors)\n\n        print(f\"F1 Score after {fold} fold is {f1_score(y_valid_vector, predictions)}\")\n\n        f1Score.append(f1_score(y_valid_vector, predictions))\n\n#         testPredictions = model.predict(testVectors)\n\n#         final_predictions.append(testPredictions)\n\n    return np.mean(f1Score)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T07:59:58.670388Z","iopub.execute_input":"2022-07-11T07:59:58.670869Z","iopub.status.idle":"2022-07-11T07:59:58.687433Z","shell.execute_reply.started":"2022-07-11T07:59:58.670831Z","shell.execute_reply":"2022-07-11T07:59:58.686282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"maximize\",study_name=\"XGBoost Hyperparameter Tuning\")\nstudy.optimize(objective, n_trials=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T07:59:59.767774Z","iopub.execute_input":"2022-07-11T07:59:59.768155Z","iopub.status.idle":"2022-07-11T13:28:35.352717Z","shell.execute_reply.started":"2022-07-11T07:59:59.768123Z","shell.execute_reply":"2022-07-11T13:28:35.348316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgboost_best_params = study.best_params\nxgboost_best_params","metadata":{"execution":{"iopub.status.busy":"2022-07-11T13:30:35.371552Z","iopub.execute_input":"2022-07-11T13:30:35.372307Z","iopub.status.idle":"2022-07-11T13:30:35.381565Z","shell.execute_reply.started":"2022-07-11T13:30:35.372259Z","shell.execute_reply":"2022-07-11T13:30:35.380243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgboost_best_params = {'learning_rate': 0.003271232395930842,\n 'reg_lambda': 3.265825567567721e-08,\n 'reg_alpha': 8.397918358018894e-06,\n 'subsample': 0.6050118955710034,\n 'colsample_bytree': 0.839587495708783,\n 'max_depth': 10,\n 'n_estimators': 17587}","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:06:23.457115Z","iopub.execute_input":"2022-07-11T16:06:23.458192Z","iopub.status.idle":"2022-07-11T16:06:23.463903Z","shell.execute_reply.started":"2022-07-11T16:06:23.458146Z","shell.execute_reply":"2022-07-11T16:06:23.462595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training XGB with tuned Parameters","metadata":{}},{"cell_type":"code","source":"f1Score = []\n\nfor fold,(train_idx,valid_idx) in enumerate(kfold.split(train_data, train_data.target)):\n    \n    X_train = train_data.iloc[train_idx][\"text\"]\n    X_valid = train_data.iloc[valid_idx][\"text\"]\n    y_train = train_data.iloc[train_idx][\"target\"]\n    y_valid = train_data.iloc[valid_idx][\"target\"]\n    \n    vectorizer = CountVectorizer(max_features=1000)\n    \n    trainVectors = vectorizer.fit_transform(X_train).toarray()\n    validVectors = vectorizer.transform(X_valid).toarray()\n    \n    y_train_vector = np.array(y_train)\n    y_valid_vector = np.array(y_valid)\n    \n    model = XGBClassifier(\n                        random_state=fold,\n                        objective='binary:logistic',\n                        tree_method='gpu_hist',  \n                        gpu_id=0,\n                        predictor='gpu_predictor',\n                        n_jobs = -1,\n                        **xgboost_best_params\n                        )\n    \n    model.fit(trainVectors,y_train_vector)\n    \n    predictions = model.predict(validVectors)\n    \n    print(f\"F1 Score after {fold} fold is {f1_score(y_valid_vector, predictions)}\")\n    \n    f1Score.append(f1_score(y_valid_vector, predictions))\n    \nprint(f\"The mean F1 Score is {np.mean(f1Score)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:06:32.647845Z","iopub.execute_input":"2022-07-11T16:06:32.648834Z","iopub.status.idle":"2022-07-11T16:18:16.081464Z","shell.execute_reply.started":"2022-07-11T16:06:32.648786Z","shell.execute_reply":"2022-07-11T16:18:16.080446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Created a simple function for predictions of test data\ndef predictions_on_test(testdata,model):\n    testData = vectorizer.transform(testdata[\"text\"]).toarray()\n    testPredictions = model.predict(testData)\n    testDataOutput = pd.DataFrame({\"id\":test_data[\"id\"],\n                                  \"target\":testPredictions})\n    return testDataOutput\n\ntestDataOutput = predictions_on_test(test_data,model)\ntestDataOutput","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:18:56.779704Z","iopub.execute_input":"2022-07-11T16:18:56.780842Z","iopub.status.idle":"2022-07-11T16:18:57.647697Z","shell.execute_reply.started":"2022-07-11T16:18:56.780765Z","shell.execute_reply":"2022-07-11T16:18:57.646710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testDataOutput.to_csv(\"fifthsubmission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-11T16:19:04.231340Z","iopub.execute_input":"2022-07-11T16:19:04.231690Z","iopub.status.idle":"2022-07-11T16:19:04.244694Z","shell.execute_reply.started":"2022-07-11T16:19:04.231661Z","shell.execute_reply":"2022-07-11T16:19:04.243619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}