{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"markdown","source":"Not so long ago there was this competition called DonorsChoose.org Application Screening. It was one of my firsts competitions in Kaggle and I learn a lot about text analysis loooking at the kernels from this competition. If you don't kno where to start I think that competition will be the first place to look up. Also you will find the kernel of the guy that won the competition, unfurtunately I haven read the code but If you are an eager learner as I am, you will use the same kernel to uderstand how and why it beat so many models in DonosChoose competition. Also I have to add that DonorsChoose is a very intuitive and dun competition so you will have agreat time exploring it! "},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold, RepeatedKFold\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom sklearn.metrics import roc_auc_score\n\nimport os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#Some basic features:\n\ntrain['question_len']   = train['question_text'].apply(lambda x: len(str(x)))\ntest['question_len']   = test['question_text'].apply(lambda x: len(str(x)))\n\ndf_all = pd.concat([train, test], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"529d41cfd03223c193ddc78b39c6d9432055c428"},"cell_type":"code","source":"print(train.shape)\nprint(train.head())\nprint(train.columns)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b6cc22b8d5ecdb0fbc61cbc84491a2c1b27561d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"133e349b581edb9f0a14018cd830a9057ecd23a7"},"cell_type":"markdown","source":"What TfidfVectorizer does?\nThis classs of sklearn allow us to compare words inside two texts and tell us how each of these words are realted within the texts. When we apply it for diferent number of texts the algorithm return a score that represents how strange is the word in relation to the other texts. In this example we can see that the word one is the most repeated and it appear in all the texts, then it must have the lowest possible value. In the other hand four only appear in the first text then it must have the higher poissible value."},{"metadata":{"trusted":true,"_uuid":"c799193464ebd3f78966a389b6b5cbac9b48bcce"},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ncorpus_3 = [\"one two three four \",\n           \"one two three  \",\n           \"one two \",\n           \"one\"]\nvectorizer = TfidfVectorizer(min_df=1)\nX = vectorizer.fit_transform(corpus_3)\nidf = vectorizer.idf_\nprint(dict(zip(vectorizer.get_feature_names(), idf)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68a235cd3f979b873537cafaaf6dd46687778f8c"},"cell_type":"code","source":"print('Preprocessing text...')\ncols = [\"question_text\"]\nn_features = [\n    400, # Number of different features for project_title\n    4040, \n    400\n]\n\nfor c_i, c in tqdm(enumerate(cols)):\n    tfidf = TfidfVectorizer(max_features=n_features[c_i], min_df=3)\n    tfidf.fit(df_all[c])\n    tfidf_train = np.array(tfidf.transform(train[c]).todense(), dtype=np.float16) \n    tfidf_test = np.array(tfidf.transform(test[c]).todense(), dtype=np.float16)\n\n    for i in range(n_features[c_i]):\n        train[c + '_tfidf_' + str(i)] = tfidf_train[:, i]\n        test[c + '_tfidf_' + str(i)] = tfidf_test[:, i]\n        \n    del tfidf, tfidf_train, tfidf_test    \nprint('Done.')\ndel df_all\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d9a96cd968265fcd21ec7a46925d14b24d6f35bd"},"cell_type":"markdown","source":"PCA"},{"metadata":{"trusted":true,"_uuid":"e8c7919e66d0fd6bd6a7b9c61577f6ecc7fb2add"},"cell_type":"code","source":"train.reset_index(drop= True)\ntest.reset_index(drop= True)\n\npca_columns=train.columns[train.columns.str.contains(\"_tfidf_\")]\ntrain_pca = train[pca_columns]\ntest_pca = test[pca_columns]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6ae46fe77e97093a84a2c8728eae25e1a697994c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"724674b67c5afc3365939627166b28af60f9878e"},"cell_type":"code","source":"train = train.drop(pca_columns, axis=1, errors='ignore')\ntest = test.drop(pca_columns, axis=1, errors='ignore')\n\ntrain = train.merge(train_pca, right_index=True, left_index = True)\ntest = test.merge(test_pca, right_index=True, left_index = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0de52a9b3fced5d9da97d9133b0f8f1c4d4e43b5"},"cell_type":"code","source":"cols_to_drop = [\"qid\", \"question_text\", \"target\"]\n\nX = train.drop(cols_to_drop, axis=1, errors='ignore')\ny = train['target']\nX_test = test.drop(cols_to_drop, axis=1, errors='ignore')\nid_test = test['qid'].values\nfeature_names = list(X.columns)\nprint(X.shape, X_test.shape)\n\n\n# Build the model\ncnt = 0\np_buf = []\nn_splits = 5\nn_repeats = 1\nkf = RepeatedKFold(\n    n_splits=n_splits, \n    n_repeats=n_repeats, \n    random_state=0)\nauc_buf = []   \n\n\nfor train_index, valid_index in kf.split(X):\n    print('Fold {}/{}'.format(cnt + 1, n_splits))\n    params = {\n          'objective': 'binary',\n          'metric': 'auc',\n          'boosting_type': 'dart',\n          'learning_rate': 0.08,\n          'max_bin': 15,\n          'max_depth': 10, #17\n          'num_leaves': 30, #63\n          'subsample': 0.8,\n          'subsample_freq': 5,\n          'colsample_bytree': 0.8,\n          'reg_lambda': 7,\n          'num_threads': 4}\n    \n    model = lgb.train(\n        params,\n        lgb.Dataset(X.loc[train_index], y.loc[train_index], feature_name=feature_names),\n        num_boost_round=10000,\n        valid_sets=[lgb.Dataset(X.loc[valid_index], y.loc[valid_index])],\n        early_stopping_rounds=100,\n        verbose_eval=100,\n    )\n\n    if cnt == 0:\n        importance = model.feature_importance()\n        model_fnames = model.feature_name()\n        tuples = sorted(zip(model_fnames, importance), key=lambda x: x[1])[::-1]\n        tuples = [x for x in tuples if x[1] > 0]\n        print('Important features:')\n        print(tuples[:50])\n\n    p = model.predict(X.loc[valid_index], num_iteration=model.best_iteration)\n    auc = roc_auc_score(y.loc[valid_index], p)\n\n    print('{} AUC: {}'.format(cnt, auc))\n\n    p = model.predict(X_test, num_iteration=model.best_iteration)\n    if len(p_buf) == 0:\n        p_buf = np.array(p)\n    else:\n        p_buf += np.array(p)\n    auc_buf.append(auc)\n\n    cnt += 1\n    if cnt > 0: # Comment this to run several folds\n        break\n    \n    del model\n    gc.collect\n\nauc_mean = np.mean(auc_buf)\nauc_std = np.std(auc_buf)\nprint('AUC = {:.6f} +/- {:.6f}'.format(auc_mean, auc_std))\n\npreds = p_buf/cnt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"45f0be34253c873b940613046bab7870bb2b6c2f"},"cell_type":"code","source":"import xgboost as xgb\nfrom xgboost import XGBRegressor\nfrom xgboost import plot_importance\ntrain = train.rename(columns = {'<lambda>':'lamda'})\ntest = test.rename(columns = {'<lambda>':'lamda'})\n\nX = train.drop(cols_to_drop, axis=1, errors='ignore')\ny = train['target']\nX_test = test.drop(cols_to_drop, axis=1, errors='ignore')\nid_test = test['qid'].values\nfeature_names = list(X.columns)\nprint(X.shape, X_test.shape)\n\n# Build the model\ncnt = 0\np_buf = []\nn_splits = 5\nn_repeats = 1\nkf = RepeatedKFold(\n    n_splits=n_splits, \n    n_repeats=n_repeats, \n    random_state=0)\nauc_buf = []   \n\n\nfor train_index, valid_index in kf.split(X):\n    print('Fold {}/{}'.format(cnt + 1, n_splits))\n    \n    xlf = XGBRegressor(\n          objective= 'binary:logistic',\n          eval_metric= 'auc',\n          eta=0.01,\n          max_depth= 7,\n          subsample =0.8, \n          colsample_bytree= 0.4,\n          min_child_weight= 10,\n          gamma= 2)\n    \n    \n    #xlf.fit(X.loc[train_index], y.loc[train_index], eval_metric='rmse')\n    xlf.fit(X.loc[train_index], y.loc[train_index], eval_metric='rmse', verbose = True, eval_set = [(X[train_index], y[train_index]), (X[valid_index], y[valid_index])], early_stopping_rounds=80)\n    p = xlf.predict(X.loc[valid_index])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae5bb7b327818bbdcd6a4b3186dbde3b11aa1179"},"cell_type":"code","source":"subm = pd.DataFrame()\nsubm['qid'] = id_test\nsubm['prediction'] = preds.round().astype(int)\nsubm.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"989ba4ebb53843929c1257eb7d2b470669f1f468"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"385e1bce9c8612a80565493701b02587f6520bbd"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}