{"cells":[{"metadata":{"_uuid":"16b8124fde7fb9469bf56a0107962a4907f8a065"},"cell_type":"markdown","source":""},{"metadata":{"_uuid":"14f9007d062a51624b85827ce7b8fbc9141e43e0"},"cell_type":"markdown","source":"# Two Sigma: Rental Interest Competition"},{"metadata":{"_uuid":"f97ea7792d3fc67eba0dc049c43709348fcb8447"},"cell_type":"markdown","source":"Note: It is likely that this kernel won't run on the Kaggle infrastructure, you can view it here: https://github.com/smspillaz/aalto-CS-kaggle-competitions/blob/master/two-sigma-rental-interest/kernel.ipynb"},{"metadata":{"trusted":false,"_uuid":"6245b63153821ec5ffc7b2e5b849d0c150d1429d"},"cell_type":"code","source":"from IPython.display import display\nimport datetime\nimport gc\nimport itertools\nimport json\nimport operator\nimport os\nimport pandas as pd\nimport pickle\nimport pprint\nimport numpy as np\nimport re\nimport seaborn as sns\nimport spacy\nimport torch\nimport torch.optim as optim\n\nfrom collections import Counter, deque\nfrom pytorch_pretrained_bert import BertAdam\nfrom sklearn.base import clone\nfrom sklearn.metrics import (\n    accuracy_score,\n    log_loss,\n    make_scorer,\n    mean_squared_error\n)\nfrom sklearn.model_selection import (\n    GridSearchCV,\n    StratifiedKFold,\n    cross_val_score,\n    train_test_split\n)\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.preprocessing import StandardScaler\nfrom skorch.callbacks import (\n    Callback,\n    Checkpoint,\n    EpochScoring,\n    LRScheduler,\n    ProgressBar,\n    TrainEndCheckpoint\n)\n\nfrom utils.callbacks import SummarizeParameters\nfrom utils.data import load_training_test_data\nfrom utils.dataframe import (\n    categories_from_column,\n    column_list_to_category_flags,\n    count_json_in_dataframes,\n    count_ngrams_up_to_n,\n    drop_columns_from_dataframes,\n    map_categorical_column_to_category_ids,\n    normalize_categories,\n    normalize_description,\n    numerical_feature_engineering_on_dataframe,\n    parse_address_components,\n    remove_outliers,\n    remap_column,\n    remap_columns_with_transform,\n    remap_date_column_to_days_before,\n    remove_small_or_stopwords_from_ranking\n)\nfrom utils.featurize import (\n    featurize_for_tabular_models,\n    featurize_for_tree_models,\n)\nfrom utils.gc import gc_and_clear_caches\nfrom utils.doc2vec import (\n    column_to_doc_vectors\n)\nfrom utils.model import (\n    basic_logistic_regression_pipeline,\n    basic_xgboost_pipeline,\n    basic_adaboost_pipeline,\n    basic_extratrees_pipeline,\n    basic_svc_pipeline,\n    basic_random_forest_pipeline,\n    expand_onehot_encoding,\n    format_statistics,\n    get_prediction_probabilities_with_columns,\n    get_prediction_probabilities_with_columns_from_predictions,\n    prediction_accuracy,\n    write_predictions_table_to_csv,\n    rescale_features_and_split_into_continuous_and_categorical,\n    split_into_continuous_and_categorical,\n    test_model_with_k_fold_cross_validation,\n    train_model_and_get_validation_and_test_set_predictions\n)\nfrom utils.language_models.bert import (\n    BertClassifier,\n    BertForSequenceClassification,\n    TensorTuple,\n    GRADIENT_ACCUMULATION_STEPS,\n    WARMUP_PROPORTION,\n    bert_featurize_data_frames,\n    create_bert_model,\n    create_bert_model_with_tabular_features,\n)\nfrom utils.language_models.descriptions import (\n    descriptions_to_word_sequences,\n    generate_bigrams,\n    generate_description_sequences,\n    maybe_cuda,\n    postprocess_sequences,\n    token_dictionary_seq_encoder,\n    tokenize_sequences,\n    torchtext_create_text_vocab,\n    torchtext_process_texts,\n    words_to_one_hot_lookups,\n)\nfrom utils.language_models.featurize import (\n    featurize_sequences_from_dataframe,\n    featurize_sequences_from_sentence_lists,\n)\nfrom utils.language_models.fasttext import (\n    FastText,\n    FastTextWithTabularData\n)\nfrom utils.language_models.simple_rnn import (\n    CheckpointAndKeepBest,\n    LRAnnealing,\n    NoToTensorInLossClassifier,\n    SimpleRNNPredictor,\n    SimpleRNNTabularDataPredictor\n)\nfrom utils.language_models.split import (\n    shuffled_train_test_split_by_indices,\n    simple_train_test_split_without_shuffle_func,\n    ordered_train_test_split_with_oversampling\n)\nfrom utils.language_models.textcnn import (\n    TextCNN,\n    TextCNNWithTabularData\n)\nfrom utils.language_models.ulmfit import (\n    load_ulmfit_classifier_with_transfer_learning_from_data_frame,\n    train_ulmfit_model_and_get_validation_and_test_set_predictions,\n    train_ulmfit_classifier_with_gradual_unfreezing\n)\nfrom utils.language_models.visualization import (\n    preview_tokenization,\n    preview_encoded_sentences\n)\nfrom utils.report import (\n    generate_classification_report_from_preds,\n    generate_classification_report\n)\n\nnlp = spacy.load(\"en\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0dadbf9422b320a526c507cfa804543dbc3ee597"},"cell_type":"markdown","source":"Check GPU support"},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"f0b96dac7233df58d513ff492b8f29de29f6f2e9"},"cell_type":"code","source":"torch.cuda.is_available()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3e9f962b884a4dde6ffe30e2b1e7d1e26deb33b8"},"cell_type":"markdown","source":"## 1 Exploratory Data Analysis"},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"4a6168216f0f76fb8579748ff156c5f46832e197"},"cell_type":"code","source":"(ALL_TRAIN_DATAFRAME, TEST_DATAFRAME) = \\\n  load_training_test_data(os.path.join('data', 'train.json'),\n                          os.path.join('data', 'test.json'))\nTRAIN_INDEX, VALIDATION_INDEX = train_test_split(ALL_TRAIN_DATAFRAME.index, test_size=0.1)\nTRAIN_DATAFRAME = ALL_TRAIN_DATAFRAME.iloc[TRAIN_INDEX].reset_index()\nVALIDATION_DATAFRAME = ALL_TRAIN_DATAFRAME.iloc[VALIDATION_INDEX].reset_index()\nTEST_DATAFRAME = TEST_DATAFRAME.reset_index(drop=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6b0c48bfae1b82070c5114f61781f723253d2036"},"cell_type":"markdown","source":"## 1.1 Initial Data Visualization\n\nTake all the numerical features and show some statistics on all of them.\n\nImmediately we can tell the following:\n - Price seems to have a pretty high range on the log scale, going from $40 \\to 449000$. That's not particularly helpful since we think this is probably a pretty important feature, so lets filter out some of the super-high priced stuff.\n - Most properties are within the sweet spot of \"-74 to -73\" longitude and \"40.70 to 40.80 latitude\". There's a few others that aren't, so maybe better to filter those out.\n - Clearly there is some bogus data. Some properties have a lat/long of 0 which is incorrect. Unfortunately this also exists on the test set, but there are probably so few that we don't care."},{"metadata":{"trusted":false,"_uuid":"ca9ec96296a3c1b55bb1561638bd838a411ec8b1"},"cell_type":"code","source":"ALL_TRAIN_DATAFRAME.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"e7053b7b895d4ffd05774b4b60658beaffcef335"},"cell_type":"code","source":"ALL_TRAIN_DATAFRAME.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"80509d6b095098d094ea80a7a61651fd1738f3b1"},"cell_type":"code","source":"TEST_DATAFRAME.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8c5ee22162f3ec9a0e9bdff1a5ac2aecfbdf5cb0"},"cell_type":"code","source":"TEST_DATAFRAME.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"bf865031052b29fd6847ab8aac732d690c883a5a"},"cell_type":"code","source":"CORE_NUMERICAL_COLUMNS = ['bathrooms', 'bedrooms', 'price', 'latitude', 'longitude']","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"da60638b1a2cfaed122d06512e228b01734f659d"},"cell_type":"code","source":"NUMERICAL_QUANTILES = {\n    'bathrooms': (0.0, 0.999),\n    'bedrooms': (0.0, 0.999),\n    'latitude': (0.01, 0.99),\n    'longitude': (0.01, 0.99),\n    'price': (0.01, 0.99)\n}","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"020ecd1f54b0c2abe069c27bbbff7bd5257fa44b"},"cell_type":"markdown","source":"We can then run a pairplot analysis on both the training and test dataframes and notice that while there isn't much correlation between price and amenities, there is quite a big correlation between things like price and location. Properties in pricier areas tend to have more amenities.\n\nThe cell for the correlation between latitude and longitude looks a little bit like the Greater New York area.\n\nPrice wise, most properties are sitting in the $2000-4000 range, steeply dropping off after that. Also, there are definitely \"pricey\" and \"cheap\" areas, with most of the cheaper properties sitting in the -73.85 to -73.95 longitude."},{"metadata":{"trusted":false,"_uuid":"dff1e7ac71fcdacf321c5502925861a94bec9c1e"},"cell_type":"code","source":"sns.pairplot(remove_outliers(ALL_TRAIN_DATAFRAME[CORE_NUMERICAL_COLUMNS],\n                             NUMERICAL_QUANTILES))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e3f240260cc6b4dc875a32694e860027fc77be6d"},"cell_type":"markdown","source":"The test set is distributed in a simialr way, though it has more properties to the north and south east."},{"metadata":{"trusted":false,"_uuid":"3047de592f0ab17f038473e7fc35d6e291a0152a"},"cell_type":"code","source":"sns.pairplot(remove_outliers(TEST_DATAFRAME[CORE_NUMERICAL_COLUMNS],\n                             NUMERICAL_QUANTILES))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"eb7aae6af2cfe9063b5ee62a397df392070d228e"},"cell_type":"markdown","source":"## 1.2 Outlier Removal"},{"metadata":{"_uuid":"e36250711504191a29943f791cea75b79ed29894"},"cell_type":"markdown","source":"Lets remove our outliers the training set."},{"metadata":{"trusted":false,"_uuid":"a63da6b4a4a39b56beada8a699d76f0dabe7857c"},"cell_type":"code","source":"TRAIN_DATAFRAME = remove_outliers(TRAIN_DATAFRAME, NUMERICAL_QUANTILES)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9ae9dd2678dd1682da49f48fb444bd567935bc36"},"cell_type":"markdown","source":"## 2 Data Cleaning and Feature Engineering"},{"metadata":{"_uuid":"81eed65b4118bdaf4e4f73381571909056652c9a"},"cell_type":"markdown","source":"Let's see what this table looks like. We'll display the head of the table which shows its features"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"c149e27d06b32cd2b6094d852e0d83d6b6a10efe"},"cell_type":"code","source":"TRAIN_DATAFRAME.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f120d8d88209fcb511adfe14dee656db6d294be8"},"cell_type":"markdown","source":"### 2.1) Cleaning up categories"},{"metadata":{"_uuid":"60e390257332b04b374079aca5f0382b200c84b8"},"cell_type":"markdown","source":"Let's clean up the categories and put them into a sensible vector. Unfortunately the categories are a bit of a mess - since the user can specify what categories they want there isn't much in the way of consistency between categories.\n\nSome of the patterns that we frequently see in the categories are:\n - Separating category names with \"**\"\n - Mix of caps/nocaps\n - Some common themes, such as:\n   - \"pets\"\n   - \"office\"\n   - \"living room\"\n   - \"garden\"\n   - \"common area\"\n   - \"storage\"\n   - \"no pets\"\n   - \"parking\"\n   - \"bicycle\"\n   - \"doorman\"\n   - etc\n\nTo deal with this, lets pull out all of the categories and normalize them\nby removing excess punctuation, normalizing for whitespace, lowercasing, and counting for certain n-grams."},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"ff0c83c3e52b9dcf3bcb782ad2f5c860f14fbe99"},"cell_type":"code","source":"normalized_categories = sorted(normalize_categories(categories_from_column(TRAIN_DATAFRAME, 'features')))\nnormalized_categories[:50]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bd28e79a8ffb9edf2bb31cd05b07636667fe1c43"},"cell_type":"markdown","source":"Now that we have our slightly tidied up categories, we can create some n-grams and count their frequency"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"0c02d7eeb388903853f1923a5b01b1e24a677fcd"},"cell_type":"code","source":"most_common_ngrams = sorted(count_ngrams_up_to_n(\" \".join(normalized_categories), 3).most_common(),\n                            key=lambda x: (-x[1], x[0]))\nmost_common_ngrams[:50]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df90556aa33b7192bf098e644bd2a5c34d46049d"},"cell_type":"markdown","source":"There's quite a few words here that don't add much value. We can remove them by consulting a list of stopwords"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"7ee88cac65ae5450ab3d24f12c32d88b836eb249"},"cell_type":"code","source":"most_common_ngrams = sorted(list(remove_small_or_stopwords_from_ranking(most_common_ngrams, nlp, 3)),\n                            key=lambda x: (-x[1], x[0]))\nmost_common_ngrams[:50]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"479dcf485c4e68d2633570d4b8a3be372082f707"},"cell_type":"markdown","source":"Now that we have these, we can probably take 100 most common and arrange\nthem into category flags for our table"},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"8f138058e3da4d8d0d0968e13bb464c481a07229"},"cell_type":"code","source":"TRAIN_DATAFRAME = column_list_to_category_flags(TRAIN_DATAFRAME, 'features', list(map(operator.itemgetter(0), most_common_ngrams[:100])))\nVALIDATION_DATAFRAME = column_list_to_category_flags(VALIDATION_DATAFRAME, 'features', list(map(operator.itemgetter(0), most_common_ngrams[:100])))\nTEST_DATAFRAME = column_list_to_category_flags(TEST_DATAFRAME, 'features', list(map(operator.itemgetter(0), most_common_ngrams[:100])))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"7bf730b06bfddf58e5aa4389240e82bda7d0741c"},"cell_type":"code","source":"TRAIN_DATAFRAME.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a46f68549199d7df1ff64e67479b14dc183ce6ea"},"cell_type":"markdown","source":"### 2.2 Cleaning up listing_date"},{"metadata":{"_uuid":"38fff8dd66dc4a5367c038d96e5caee9788dfc70"},"cell_type":"markdown","source":"We can also do something useful with the listing date - it may be better to say how many days ago the property was listed - older properties are probably going to get a lot less interest than newer properties."},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"d3b5ae5066a8f5cb4ce81cb1922699643026262a"},"cell_type":"code","source":"TRAIN_DATAFRAME = remap_date_column_to_days_before(TRAIN_DATAFRAME, \"created\", \"created_days_ago\", datetime.datetime(2017, 1, 1))\nVALIDATION_DATAFRAME = remap_date_column_to_days_before(VALIDATION_DATAFRAME, \"created\", \"created_days_ago\", datetime.datetime(2017, 1, 1))\nTEST_DATAFRAME = remap_date_column_to_days_before(TEST_DATAFRAME, \"created\", \"created_days_ago\", datetime.datetime(2017, 1, 1))","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"292cdc492a6ab5697651c16cc52b05aabe932ee0"},"cell_type":"code","source":"TRAIN_DATAFRAME[\"created_days_ago\"].head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0866a680ef73635676af0e5252cf92cda000212d"},"cell_type":"markdown","source":"### 2.3 Cleaning up interest_level"},{"metadata":{"_uuid":"136c21ff0c22cf9529abed95b15091b6e0d5b531"},"cell_type":"markdown","source":"Right now the interest level is encoded on a scale of \"Low, Medium, High\". The competition\nwants us to classify the entries in to each, so we assign a label"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"44e40170a6ee975f5c6a11a0afeb2d3d2bdad37a"},"cell_type":"code","source":"INTEREST_LEVEL_MAPPINGS = {\n    \"high\": 0,\n    \"medium\": 1,\n    \"low\": 2\n}\n\nTRAIN_DATAFRAME = remap_column(TRAIN_DATAFRAME, \"interest_level\", \"label_interest_level\", lambda x: INTEREST_LEVEL_MAPPINGS[x])\nVALIDATION_DATAFRAME = remap_column(VALIDATION_DATAFRAME, \"interest_level\", \"label_interest_level\", lambda x: INTEREST_LEVEL_MAPPINGS[x])\n# The TEST_DATAFRAME does not have an interest_level column, so we\n# instead add it and replace it with all zeros\nTEST_DATAFRAME[\"label_interest_level\"] = 0","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"1ef3a07dca674f6ad930ef7f159d96b27a992542"},"cell_type":"code","source":"TRAIN_DATAFRAME[\"label_interest_level\"].head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"482eb5f7383baec56fabf7b60366a1fc27d7dede"},"cell_type":"markdown","source":"### 2.4 Cleaning up building_id, manager_id"},{"metadata":{"_uuid":"eacdff2047700cedf8008cdce2520d3db3083ae4"},"cell_type":"markdown","source":"`building_id` and `manager_id` look a bit useless to us on the outside, but according to https://www.kaggle.com/den3b81/some-insights-on-building-id they are actually quite predictive of interest since 20% of the manager make up 80% of the rentals (we can also see this in their writing style as well).\n\nSince there aren't too many managers or buildings in total, we can convert these into category ID's where we'll pass them through an embedding later on.\n\nNote that we need to do this over both dataframes - since there could\nbe some managers that are in the test dataframe which are not in the training dataframe and vice versa.\n\nNote that we want to lump all the \"misc\" buildings and managers together\ninto a single building or manager since listings by \"non-property managers\" or \"non-frequently-rented-buildings\" are different from ones run by property managers."},{"metadata":{"trusted":false,"_uuid":"38c4e06fc99eca92c21d196bc4b761ada51759f6"},"cell_type":"code","source":"((BUILDING_ID_UNKNOWN_REMAPPING,\n  BUILDING_CATEGORY_TO_BUILDING_ID,\n  BUILDING_CATEGORY_TO_BUILDING_ID),\n (TRAIN_DATAFRAME,\n  VALIDATION_DATAFRAME,\n  TEST_DATAFRAME)) = map_categorical_column_to_category_ids(\n    'building_id',\n    'building_id_category',\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n    min_freq=40\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"74e2f78a868f2199464613f4c47e314fb42026b6"},"cell_type":"code","source":"((MANAGER_ID_UNKNOWN_REMAPPING,\n  MANAGER_ID_TO_MANAGER_CATEGORY,\n  MANAGER_CATEGORY_TO_MANAGER_ID),\n (TRAIN_DATAFRAME,\n  VALIDATION_DATAFRAME,\n  TEST_DATAFRAME)) = map_categorical_column_to_category_ids(\n    'manager_id',\n    'manager_id_category',\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n    min_freq=40\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27f0836e1cb9e6397906f4e8931a9d81c0022b1d"},"cell_type":"markdown","source":"### 2.5 Parsing and Separating Out Address Components\nSome properties might be in the same neighbourhood, the same street or\npart of the same building. If we separate out the address components then\nwe might be able to get some more meaningful feature groupings.\n\nWe first parse all the components into their own columns and then map them into categories (dropping them later on)."},{"metadata":{"trusted":false,"_uuid":"5fa97d2b894320da3ff1ffc48e2246118f441331"},"cell_type":"code","source":"(TRAIN_DATAFRAME,\n VALIDATION_DATAFRAME,\n TEST_DATAFRAME) = parse_address_components(\n    [\n        \"display_address\",\n        \"street_address\"\n    ],\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c1d4dea51b43215c5bb75b15636d8926eb3c0079"},"cell_type":"code","source":"((DISP_ADDR_ID_UNKNOWN_REMAPPING,\n  DISP_ADDR_ID_TO_DISP_ADDR_CATEGORY,\n  DISP_ADDR_CATEGORY_TO_DISP_ADDR_ID),\n (TRAIN_DATAFRAME,\n  VALIDATION_DATAFRAME,\n  TEST_DATAFRAME)) = map_categorical_column_to_category_ids(\n    'display_address_normalized',\n    'display_address_category',\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n    min_freq=40\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"90198c9ec3464d1e4b9d41e0ed9e889e7481c517"},"cell_type":"markdown","source":"### 2.6 Counting Number of Photos\nThe number of photos a place has might be predictive of its interest as well, so lets at least count the number of photos."},{"metadata":{"trusted":false,"_uuid":"5f5c9fb264c63de1cc9bd501d0fde5e42f827b2e"},"cell_type":"code","source":"(TRAIN_DATAFRAME,\n VALIDATION_DATAFRAME,\n TEST_DATAFRAME) = count_json_in_dataframes(\n    \"photos\",\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b661a16758446836cc134117751ae2f560734e9d"},"cell_type":"markdown","source":"### 2.7 Feature Engineering on Numerical Columns\nSome models can't do simple math, but ratios or additions/subtractions\nbetween things might be important. Lets do that now for all of our\nnumerical data, but only for XGBoost, later"},{"metadata":{"trusted":false,"_uuid":"00faf3bd75480a3bdb829a9624a859ffa6c9125e"},"cell_type":"code","source":"NUMERICAL_COLUMNS = CORE_NUMERICAL_COLUMNS + [\n    'photos_count'\n]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7a62b2139b8221e58b1828fe79beb86d73532191"},"cell_type":"markdown","source":"### 2.8 Cleaning Description\nThe text in the descriptions are pretty messy. We can clean it up by applying some normalization (eg, removing repeated symbols, normalizing whitespace, etc)"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"01ba018298e533428f8f7eb28b1eae3f525e6fab"},"cell_type":"code","source":"(TRAIN_DATAFRAME,\n VALIDATION_DATAFRAME,\n TEST_DATAFRAME) = remap_columns_with_transform(\n    'description',\n    'clean_description',\n    normalize_description,\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME,\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb2e19cf3f25652047cf7cd09e61f92aa1b2fab5"},"cell_type":"markdown","source":"### 2.9 Drop unnecessary columns"},{"metadata":{"_uuid":"4126b6adf99d5db57a34dae201c2d80bf24731f2"},"cell_type":"markdown","source":"Now that we have made our data nicer to work with, we can drop all the text-only columns and keep a \"features\" dataset, eg one that can be fed into our models (with a little extra work)"},{"metadata":{"trusted":false,"_uuid":"c3f94fd73dbf999b2686b117a64be9d6dcc62935"},"cell_type":"code","source":"DROP_COLUMNS = [\n    'id',\n    'index',\n    'created',\n    'building_id',\n    'clean_description',\n    'description',\n    'features',\n    'display_address',\n    'display_address_normalized',\n    # We keep listing_id in the dataframe\n    # since we'll need it later\n    # 'listing_id',\n    'manager_id',\n    'photos',\n    'street_address',\n    'street_address_normalized',\n    'interest_level',\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"1a85fd01bf92d35b79e152431bb22856c1af8436"},"cell_type":"code","source":"(FEATURES_TRAIN_DATAFRAME,\n FEATURES_VALIDATION_DATAFRAME,\n FEATURES_TEST_DATAFRAME) = drop_columns_from_dataframes(\n    DROP_COLUMNS,\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a967e3ed3faa9727a02cec25bfff406cb6a4719b"},"cell_type":"code","source":"FEATURES_TRAIN_DATAFRAME.head(5)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"dcad988ee6f9f066dfba67e7bb432bee72c57fff"},"cell_type":"markdown","source":"### 2.10 Visualizing Again\n\nWe can do the pairplots again now that we have featurized a little bit more and also to validate that our train/test split makes sense.\n\nIts hard to visualize categorical features in a pairplot so we don't do that. Instead, we visualize the photos_count and label_interest level to see if any feature in particular is highly correlated with the interest level.\n\nWe find that at least on both the training and validation data, nothing *really* is, except perhaps the latitude and longitude, indicating that location seems to be the most important factor in determining how much interest a property gets.\n\nAlso, the numbert of photos is *negatively* correlated with interest (properties with more photos) had less interest overall.\n\nThe bottom right hand corner of the pairplot matrix tells us something particularly useful which is the class (im)balance. We have lots of properties with low interest (2) and few properties with high interest. This will be a challenge for us to deal with later. We tried over-sampling by duplicating but that didn't really help validation scores at all. There are also other oversampling techniques like SMOTE but they don't work with non-numeric data."},{"metadata":{"trusted":false,"_uuid":"1974c87aaa71b96b991a83a7bc937e0ffbd1e30b"},"cell_type":"code","source":"FEATURIZED_NUMERICAL_COLUMNS = CORE_NUMERICAL_COLUMNS + [\"photos_count\", \"label_interest_level\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"242ef857622e93497dcc4cb0620783b21a5f2360"},"cell_type":"code","source":"sns.pairplot(remove_outliers(FEATURES_TRAIN_DATAFRAME[FEATURIZED_NUMERICAL_COLUMNS],\n                             NUMERICAL_QUANTILES))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"83b5e0cf38aef1b85606d9be995591a4011037d4"},"cell_type":"markdown","source":"We compare the validation split to the test data above to ensure that that we're sampling from a similar distribution. On the whole, we appear to be - the distribution of labels is about the same and this seems to have leaked into the distribution of proeprties geogrpahically (though the test data seems to have more properties from the northeastern peninsula). With a bit more time we probably could have tried to address this problem by also stratifying the validation split so that we included properties from that region, too."},{"metadata":{"trusted":false,"_uuid":"01c33e7ea83585af03e4a75a3f3e4dcbd9d1db20"},"cell_type":"code","source":"sns.pairplot(remove_outliers(FEATURES_VALIDATION_DATAFRAME[FEATURIZED_NUMERICAL_COLUMNS],\n                             NUMERICAL_QUANTILES))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1686c3f1c945b59854ad78e244bd511c8f99ac78"},"cell_type":"markdown","source":"Finally, test set: we don't have the labels here (obviously), but we do have the count of photos. We get similar-ish distributions, properties with a higher photo count tend to be a little more expensive, but not much new information there."},{"metadata":{"trusted":false,"_uuid":"7f1df91a82297054abd2ba65b9e9af6b63878398"},"cell_type":"code","source":"sns.pairplot(remove_outliers(FEATURES_TEST_DATAFRAME[FEATURIZED_NUMERICAL_COLUMNS[:-1]],\n                             NUMERICAL_QUANTILES))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27c3fa47eddf92f54b282ac013a44c152bdd0d5e"},"cell_type":"markdown","source":"## 3) Fitting models"},{"metadata":{"_uuid":"96ca318fe2bea39a2d3b41c7c50b8d20da582028"},"cell_type":"markdown","source":"Now we can try out a few models and see what works well for the data that\nwe have so far."},{"metadata":{"trusted":false,"_uuid":"3a7aadf710e2300de2237d74bbe5b07c3c075a7e"},"cell_type":"code","source":"CATEGORICAL_FEATURES = {\n    'building_id_category': len(BUILDING_CATEGORY_TO_BUILDING_ID),\n    'manager_id_category': len(MANAGER_ID_TO_MANAGER_CATEGORY),\n    'display_address_category': len(DISP_ADDR_ID_TO_DISP_ADDR_CATEGORY)\n}","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7c7add54f7cfbb843edb7dbaf1e383524cd0a41"},"cell_type":"markdown","source":"For convenience, pull out the labels for training"},{"metadata":{"trusted":false,"_uuid":"3f9d0d151379d8e8ebc7584a63e703532a47dc4d"},"cell_type":"code","source":"TRAIN_LABELS = FEATURES_TRAIN_DATAFRAME['label_interest_level']\nVALIDATION_LABELS = FEATURES_VALIDATION_DATAFRAME['label_interest_level']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"81acb587b2a2e68790a96b17df5102385d1f6859"},"cell_type":"markdown","source":"### 3.1 Logistic Regression\n\nThis is just baseline Logistic Regression. The 'C' parameter is a regularization strength.\n\nThe 'penalty' paramter specifies the regularization penalty to be applied. 'l2' is the default, which basically\nprevents any one particular weight from getting too large. 'l1' promotes sparse solutions."},{"metadata":{"trusted":false,"_uuid":"f7e657c5a8943799ed3c08c5c662bb46ef5987da"},"cell_type":"code","source":"def train_logistic_regression_model(data_info,\n                                    featurized_train_data,\n                                    featurized_validation_data,\n                                    train_labels,\n                                    validation_labels,\n                                    train_param_grid_optimal=None):\n    pipeline = basic_logistic_regression_pipeline(featurized_train_data,\n                                                  train_labels,\n                                                  CATEGORICAL_FEATURES,\n                                                  param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline\n\n\ndef predict_with_sklearn_estimator(model, data):\n    return model.predict(data), model.predict_proba(data)\n","execution_count":null,"outputs":[]},{"metadata":{"scrolled":true,"trusted":false,"_uuid":"2ed33cd48eb1d57238257ee73d0efa56418a5f6c"},"cell_type":"code","source":"(LOGISTIC_REGRESSION_MODEL_VALIDATION_PROBABILITIES,\n LOGISTIC_REGRESSION_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tabular_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_logistic_regression_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'C': [1.0],\n            'class_weight': [None],\n            'penalty': ['l2']\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ed24c9c3a933abc3475abc8a7394e5256b5cc561"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        LOGISTIC_REGRESSION_MODEL_TEST_PROBABILITIES,\n    ),\n    'renthop_logistic_regression_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"515bf2978990cb92edbfcf16bf54cfc4d62dd6e9"},"cell_type":"markdown","source":"### 3.2 XGBoost\n\nXGBoost is a histogram based model that applies boosting to an ensemble of weaker learners (decision trees). It generally performs quite well on Kaggle competitions and is also another baseline to use."},{"metadata":{"trusted":false,"_uuid":"ea30a9027309f215c29bfb372c570c98479b7dcd"},"cell_type":"code","source":"def train_xgboost_model(data_info,\n                        featurized_train_data,\n                        featurized_validation_data,\n                        train_labels,\n                        validation_labels,\n                        train_param_grid_optimal=None):\n    pipeline = basic_xgboost_pipeline(featurized_train_data,\n                                      train_labels,\n                                      tree_method=(\n                                          # 'gpu_hist' turned out to be a lot slower\n                                         'hist'\n                                      ),\n                                      param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"102ea2ed7c256b3fd880d917bfd848ef9afda3ba"},"cell_type":"code","source":"(XGBOOST_MODEL_VALIDATION_PROBABILITIES,\n XGBOOST_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tree_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_xgboost_model,\n        predict_with_sklearn_estimator,\n        # Determined by Grid Search, above\n        train_param_grid_optimal={\n            'colsample_bytree': [1.0],\n            'gamma': [1.5],\n            'max_depth': [5],\n            'min_child_weight': [1],\n            'n_estimators': [200],\n            'subsample': [0.6]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"0870a18f5fe664034a0d03320b8f22636e997b6a"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        XGBOOST_MODEL_TEST_PROBABILITIES,\n    ),\n    'xgboost_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d2d5049c72049124fe85ce1f49930d9df3a5aafb"},"cell_type":"markdown","source":"## 3.3 Random Forest\n\nRandom forest is basically an ensemble of lots of decision trees and we average out the results from each tree."},{"metadata":{"trusted":false,"_uuid":"c62a2e9af1e69d7c6f2ab53b6b2d2bd49d1e7249"},"cell_type":"code","source":"def train_rf_model(data_info,\n                   featurized_train_data,\n                   featurized_validation_data,\n                   train_labels,\n                   validation_labels,\n                   train_param_grid_optimal=None):\n    pipeline = basic_random_forest_pipeline(featurized_train_data,\n                                            train_labels,\n                                            # Determined by Grid Search, above\n                                            param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"eebd1a128ea5e23070ccdc2b9f88a19b3f90421f"},"cell_type":"code","source":"(RF_MODEL_VALIDATION_PROBABILITIES,\n RF_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tree_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_rf_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'bootstrap': [False],\n            'max_depth': [5],\n            'min_samples_leaf': [1],\n            'n_estimators': [100]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a9ee1cd39c0251ffab2223352fcf1a8ce072cbe5"},"cell_type":"markdown","source":"## 3.4 Adaboost\n\nIn this approach, we combine several \"weak\" classifiers into a \"strong\" classifier. It is actually a meta-learning technique, though we use a decision tree as the base model."},{"metadata":{"trusted":false,"_uuid":"96e66ca6596d04890cc69825f9035d88fe7fa903"},"cell_type":"code","source":"def train_adaboost_model(data_info,\n                         featurized_train_data,\n                         featurized_validation_data,\n                         train_labels,\n                         validation_labels,\n                         train_param_grid_optimal=None):\n    pipeline = basic_adaboost_pipeline(featurized_train_data,\n                                       train_labels,\n                                       # Determined by Grid Search, above\n                                       param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"77855496400b218d690f84042df3141e2b6b7531"},"cell_type":"code","source":"(ADABOOST_MODEL_VALIDATION_PROBABILITIES,\n ADABOOST_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tree_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_adaboost_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'learning_rate': [0.1],\n            'n_estimators': [100]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4f35dea71407866caea3dbff25e66769e27b1308"},"cell_type":"markdown","source":"## 3.4 ExtraTrees\n\nExtratrees is also another ensembling based model. It stands for \"Extremely Randomized Trees\". Its main property is that it reduces variance for a small incrase in bias."},{"metadata":{"trusted":false,"_uuid":"f3a218a00ec7bd94621de754e3ba6b0dff5b5a1a"},"cell_type":"code","source":"def train_extratrees_model(data_info,\n                           featurized_train_data,\n                           featurized_validation_data,\n                           train_labels,\n                           validation_labels,\n                           train_param_grid_optimal=None):\n    pipeline = basic_extratrees_pipeline(featurized_train_data,\n                                         train_labels,\n                                         # Determined by Grid Search, above\n                                         param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b606bc7176aa2fde3dca34d2d7cc01a841bd4c49"},"cell_type":"code","source":"(EXTRATREES_MODEL_VALIDATION_PROBABILITIES,\n EXTRATREES_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tree_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_extratrees_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'bootstrap': [False],\n            'max_depth': [5],\n            'min_samples_leaf': [1],\n            'n_estimators': [200]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8e59d6bcf5c4be4ea0ca84aaef16294c7a2e4af6"},"cell_type":"markdown","source":"### 3.5 SVC\n\nWith Support Vector Machines we can make use of kernel functions in order to try and have non-linear fits on our data. For instance, below we use the radial-basis-function kernel, though it doesn't perform as well as we would like."},{"metadata":{"trusted":false,"_uuid":"bad38d75802e507ec88a4a28c5e1a6d92b06c0b1"},"cell_type":"code","source":"def train_svc_model(data_info,\n                    featurized_train_data,\n                    featurized_validation_data,\n                    train_labels,\n                    validation_labels,\n                    train_param_grid_optimal=None):\n    pipeline = basic_svc_pipeline(featurized_train_data,\n                                  train_labels,\n                                  # Determined by Grid Search, above\n                                  param_grid_optimal=train_param_grid_optimal)\n    pipeline.fit(featurized_train_data, train_labels)\n    print(\"Best parameters {}\".format(pipeline.best_params_))\n    return pipeline","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"5ce42266d2647bd62212bf4301d8f9dfaa969f39"},"cell_type":"code","source":"(SVC_MODEL_VALIDATION_PROBABILITIES,\n SVC_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_for_tabular_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_svc_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'C': [1.0],\n            'gamma': ['scale'],\n            'kernel': ['rbf']\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f6b793d863eb32af85b1cf93597cb84a9087051b"},"cell_type":"markdown","source":"### 3.6 Neural Net (Text Classification) Approaches"},{"metadata":{"_uuid":"6027e7ff61790e357fba27018da9801fcc93486f"},"cell_type":"markdown","source":"Here we leverage the actual text in the \"description\" field to try and do the classification, both independently and on top of the tabular data using Neural Nets with PyTorch.\n\nBefore we do that, it will be convenient to split our\ndata into continuous and categorical sections (as the categorical\nsections will be put through independent embeddings in each\nmodel) and tensorify some of our data, so lets do that now.\n\nNote that the continuous data needs to be scaled to have zero mean\nand unit variance - this prevents saturation of activation units\nin the upper classification layers."},{"metadata":{"trusted":false,"_uuid":"089b270b3c60755e6bb256af61e5605ec8138bea"},"cell_type":"code","source":"((TRAIN_FEATURES_CONTINUOUS,\n  TRAIN_FEATURES_CATEGORICAL),\n (VALIDATION_FEATURES_CONTINUOUS,\n  VALIDATION_FEATURES_CATEGORICAL),\n (TEST_FEATURES_CONTINUOUS,\n  TEST_FEATURES_CATEGORICAL)) = rescale_features_and_split_into_continuous_and_categorical(CATEGORICAL_FEATURES,\n                                                                                           FEATURES_TRAIN_DATAFRAME,\n                                                                                           FEATURES_VALIDATION_DATAFRAME,\n                                                                                           FEATURES_TEST_DATAFRAME)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf2a25a69f9c06707972fc9d4a060bd4a0b66b0f"},"cell_type":"markdown","source":"Now we convert the labels into tensors, but we only put the labels\non the GPU (not the rest of the data, we'd run out of memory). Skorch\nwill conveniently put each batch on the GPU for us, so we don't need to\nworry about that."},{"metadata":{"trusted":false,"_uuid":"4272a0f59944aa36a5609ba972f1634edff2117c"},"cell_type":"code","source":"TRAIN_LABELS_TENSOR = torch.tensor(TRAIN_LABELS.values).long()\nVALIDATION_LABELS_TENSOR = torch.tensor(VALIDATION_LABELS.values).long()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f7e0a1481db52f2b8823cd63164eb4dea8bb10e5"},"cell_type":"code","source":"TRAIN_FEATURES_CONTINUOUS_TENSOR = torch.tensor(TRAIN_FEATURES_CONTINUOUS).float()\nTRAIN_FEATURES_CATEGORICAL_TENSOR = torch.tensor(TRAIN_FEATURES_CATEGORICAL).long()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b853b1639deffcfece52c530d13a5a839a1b6e9d"},"cell_type":"code","source":"VALIDATION_FEATURES_CONTINUOUS_TENSOR = torch.tensor(VALIDATION_FEATURES_CONTINUOUS).float()\nVALIDATION_FEATURES_CATEGORICAL_TENSOR = torch.tensor(VALIDATION_FEATURES_CATEGORICAL).long()","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"8b2043040b7c2d4190401000f9d99c9e870ffaae"},"cell_type":"code","source":"TEST_FEATURES_CONTINUOUS_TENSOR = torch.tensor(TEST_FEATURES_CONTINUOUS).float()\nTEST_FEATURES_CATEGORICAL_TENSOR = torch.tensor(TEST_FEATURES_CATEGORICAL).long()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6862b80973100c8e906c2060be70cd3428d82196"},"cell_type":"markdown","source":"#### 3.6.1 Simple RNN/LSTM"},{"metadata":{"_uuid":"e61390f42d851a00ce3a2a12ed22e8cfe7fcf7ac"},"cell_type":"markdown","source":"Before we start putting our data into the RNN, lets tokenize our descriptions. In order to do this we'll be using fastai's Tokenizer class. We can preview the result of tokenization below"},{"metadata":{"trusted":false,"_uuid":"a823cc1430f2bc1da80a9413a32dac8cfc2b8557"},"cell_type":"code","source":"preview_tokenization(TRAIN_DATAFRAME[\"description\"][:10])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b63c60d1fc18044e083082dea6588b816f3755f2"},"cell_type":"markdown","source":"Now we need to encode our tokens as **padded** sequences of integers such that we have a matrix of length [n $\\times$ max_len].\n\nTo start our with, PyTorch takes one-hot encoded data as a single array of integers, where each integer specifies the index into some sparse vector where a $1$ will be set. Of course, if you have such sparse vectors, you can save a lot of time on the multiplication by just picking the right dimension and ignoring all the zero ones, which is exactly what happens internally.\n\nThe reason why we need padding is for computational efficiency reasons - we want to push a large batch of sentences on to the GPU for parallel computation all, but in order for this to work we need to pass the GPU a big square matrix. This means that the matrix will have at least as many columns as the maximum number of tokens in a sentence, where every other shorter sentence will be padded by a special \"<PAD>\" token. We also keep the length of every unpadded sentence in a separate vector - we'll see later that this is used by torch as an optimization to prevent the RNN from running over all the padding tokens within a batch."},{"metadata":{"trusted":false,"_uuid":"4972c641c57b0b86b5a06f00b01c1aa1565f5e30"},"cell_type":"code","source":"preview_encoded_sentences(TRAIN_DATAFRAME[\"description\"][:10])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c9e7abdb02e523edf3be8988e939b4974604174c"},"cell_type":"markdown","source":"Here we roll our own simple LSTM classifier using Skorch and PyTorch.\n\nThe `SimpleRNNPredictor` model just takes a batch of encoded word-encoded\nsentences and runs them through a Bi-LSTM, then has a fully connected\nlayer sitting on top of the final \"hidden\" output of the LSTM (think\nof the hidden state being passed through the LSTM along with each encoded input for the entire sentence all the way up to the sentence length). Then we just predict the class based on the final hidden state.\n\nWhat Skorch does here is it implements the training loop, allows us\nto add \"hooks\" (for instance, scoring on every epoch, a progress bar,\ncheckpointing so that we only keep our best model by validation loss,\ncyclic learning rate scheduling and learning rate annealing (we reduce the learning rate if our validation loss is not going down)). It also wraps the model in an sklearn estimator-like API, so we can use it just like any other estimator.\n\nNote that the NoToTensorInLossClassifier implements a few fixes on top\nof NeuralNetClassifier - in particular `predict_proba` takes the exponent of the returned probabilities since we return log-softmax probabilities that get passed to `NLLLoss`."},{"metadata":{"trusted":false,"_uuid":"2904f0fdf9d20d1d33aae071efc2178631bfb22b"},"cell_type":"code","source":"def featurize_for_rnn_language_model(*dataframes):\n    data_info, model_datasets = featurize_sequences_from_dataframe(*dataframes)\n    return data_info, model_datasets\n\n\ndef train_rnn_model(data_info,\n                    featurized_train_data,\n                    featurized_validation_data,\n                    train_labels,\n                    validation_labels,\n                    train_param_grid_optimal=None):\n    word_to_one_hot, one_hot_to_word = data_info\n    train_word_description_sequences, train_word_sequences_lengths = featurized_train_data\n    model = NoToTensorInLossClassifier(\n        SimpleRNNPredictor,\n        module__encoder_dimension=100, # Number of encoder features\n        module__hidden_dimension=50, # Number of hidden features\n        module__dictionary_dimension=len(one_hot_to_word), # Dictionary dimension\n        module__output_dimension=3,\n        module__dropout=0.1,\n        lr=1e-2,\n        batch_size=256,\n        optimizer=optim.Adam,\n        max_epochs=4,\n        module__layers=2,\n        train_split=simple_train_test_split_without_shuffle_func(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='rnn_lang_checkpoint'),\n            TrainEndCheckpoint(dirname='rnn_lang_checkpoint',\n                               fn_prefix='rnn_train_end_')\n        ]\n    )\n    model.fit((train_word_description_sequences,\n               train_word_sequences_lengths),\n              maybe_cuda(train_labels))\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"82e8df802706a7db22cf0eb31b617ac5e29f2e88"},"cell_type":"code","source":"(SIMPLE_RNN_MODEL_VALIDATION_PROBABILITIES,\n SIMPLE_RNN_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_for_rnn_language_model,\n        train_rnn_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"32067621cd42756be2cdf79de9023a81b44aaf8f"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        SIMPLE_RNN_MODEL_TEST_PROBABILITIES,\n    ),\n    'simple_rnn_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f53a7c1719feda2b4dfd3aa91d2d3d008f5041c3"},"cell_type":"code","source":"def splice_into_datasets(datasets, append):\n    return tuple(list(d) + list(a) for d, a in zip(datasets, append))\n\ndef featurize_for_rnn_tabular_model(*dataframes):\n    data_info, model_datasets = featurize_sequences_from_dataframe(*dataframes)\n    # Need to wrap each dataset in outer list so that splicing works correctly\n    return data_info, splice_into_datasets(model_datasets,\n                                           ((TRAIN_FEATURES_CONTINUOUS_TENSOR,\n                                             TRAIN_FEATURES_CATEGORICAL_TENSOR),\n                                            (VALIDATION_FEATURES_CONTINUOUS_TENSOR,\n                                             VALIDATION_FEATURES_CATEGORICAL_TENSOR),\n                                            (TEST_FEATURES_CONTINUOUS_TENSOR,\n                                             TEST_FEATURES_CATEGORICAL_TENSOR )))\n\n\ndef train_rnn_tabular_model(data_info,\n                            featurized_train_data,\n                            featurized_validation_data,\n                            train_labels,\n                            validation_labels,\n                            train_param_grid_optimal=None):\n    word_to_one_hot, one_hot_to_word = data_info\n    _, _, train_continuous, train_categorical = featurized_train_data\n    model = NoToTensorInLossClassifier(\n        SimpleRNNTabularDataPredictor,\n        module__encoder_dimension=100, # Number of encoder features\n        module__hidden_dimension=50, # Number of hidden features\n        module__dictionary_dimension=len(one_hot_to_word), # Dictionary dimension\n        module__output_dimension=3,\n        module__dropout=0.1,\n        module__continuous_features_dimension=train_continuous.shape[1],\n        module__categorical_feature_embedding_dimensions=[\n            (CATEGORICAL_FEATURES[c], 80) for c in CATEGORICAL_FEATURES\n        ],\n        lr=1e-2,\n        batch_size=256,\n        optimizer=optim.Adam,\n        max_epochs=4,\n        module__layers=2,\n        train_split=simple_train_test_split_without_shuffle_func(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='rnn_lang_checkpoint'),\n            TrainEndCheckpoint(dirname='rnn_lang_checkpoint',\n                               fn_prefix='rnn_train_end_')\n        ]\n    )\n    model.fit(featurized_train_data, maybe_cuda(train_labels))\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"e89befcf41c8e384fed51a4df9d494bc2ef1ec11"},"cell_type":"code","source":"(SIMPLE_RNN_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n SIMPLE_RNN_TABULAR_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_for_rnn_tabular_model,\n        train_rnn_tabular_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"075e8a99720748d8fb13b79605452ea4501bb289"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        SIMPLE_RNN_TABULAR_MODEL_TEST_PROBABILITIES,\n    ),\n    'simple_rnn_model_tabular_data_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5a1c0e134dcc048797c35b92e39aa7c2cae8daef"},"cell_type":"markdown","source":"### 3.6.1 FastText"},{"metadata":{"_uuid":"195405c3f3b16bd4b80c58894bd31c83f2645e84"},"cell_type":"markdown","source":"A key concept here is that we compute the bigrams of a sentence and then apply them to the end of the sentence.\n\nThen we just put the whole thing through linear layers after average pooling, the average pooling takes into account the big-grams."},{"metadata":{"trusted":false,"_uuid":"39089250fb06f69e5ad5c7f98526caae8dce60b2"},"cell_type":"code","source":"def featurize_dataframe_sequences_for_fasttext(*dataframes):\n    sequences_lists = tokenize_sequences(\n        *tuple(list(df['clean_description']) for df in dataframes)\n    )\n\n    text = torchtext_create_text_vocab(*sequences_lists,\n                                       vectors='glove.6B.100d')\n    \n    return text, torchtext_process_texts(*postprocess_sequences(\n        *sequences_lists,\n        postprocessing=generate_bigrams\n    ), text=text)\n\n\ndef train_fasttext_model(data_info,\n                         featurized_train_data,\n                         featurized_validation_data,\n                         train_labels,\n                         validation_labels,\n                         train_param_grid_optimal=None):\n    embedding_dim = 100\n    model = NoToTensorInLossClassifier(\n        FastText,\n        lr=0.001,\n        batch_size=256,\n        optimizer=optim.Adam,\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='fasttext_checkpoint'),\n            TrainEndCheckpoint(dirname='fasttext_tabular_checkpoint',\n                               fn_prefix='fasttext_train_end_')\n        ],\n        max_epochs=6,\n        train_split=shuffled_train_test_split_by_indices(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        module__encoder_dimension=embedding_dim, # Number of encoder features\n        module__dictionary_dimension=len(data_info.vocab.itos), # Dictionary dimension\n        module__output_dimension=3,\n        module__dropout=0.8,\n        module__pretrained=data_info.vocab.vectors\n    )\n    model.fit(featurized_train_data, maybe_cuda(train_labels))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"3081cb2276d66d253339aa0e70bd1562142573ef"},"cell_type":"code","source":"(FASTTEXT_MODEL_VALIDATION_PROBABILITIES,\n FASTTEXT_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_dataframe_sequences_for_fasttext,\n        train_fasttext_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"b6c2e7b24ac198f056c98621fdac6c6bea6ebdd8"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        FASTTEXT_MODEL_TEST_PROBABILITIES,\n    ),\n    'fasttext_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d8913dd991c5d09e1a2d7e7f928d7fb1619d5d49"},"cell_type":"code","source":"def featurize_for_fasttext_tabular_model(*dataframes):\n    data_info, model_datasets = featurize_dataframe_sequences_for_fasttext(*dataframes)\n    # Need to wrap each dataset in outer list so that splicing works correctly\n    return data_info, splice_into_datasets(tuple([x] for x in model_datasets),\n                                           ((TRAIN_FEATURES_CONTINUOUS_TENSOR,\n                                             TRAIN_FEATURES_CATEGORICAL_TENSOR),\n                                            (VALIDATION_FEATURES_CONTINUOUS_TENSOR,\n                                             VALIDATION_FEATURES_CATEGORICAL_TENSOR),\n                                            (TEST_FEATURES_CONTINUOUS_TENSOR,\n                                             TEST_FEATURES_CATEGORICAL_TENSOR)))\n\n\ndef train_fasttext_tabular_model(data_info,\n                                 featurized_train_data,\n                                 featurized_validation_data,\n                                 train_labels,\n                                 validation_labels,\n                                 train_param_grid_optimal=None):\n    embedding_dim = 100\n    _, train_features_continuous, _ = featurized_train_data\n    model = NoToTensorInLossClassifier(\n        FastTextWithTabularData,\n        lr=0.001,\n        batch_size=256,\n        optimizer=optim.Adam,\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='fasttext_checkpoint'),\n            TrainEndCheckpoint(dirname='fasttext_tabular_checkpoint',\n                               fn_prefix='fasttext_train_end_')\n        ],\n        max_epochs=6,\n        train_split=shuffled_train_test_split_by_indices(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        module__encoder_dimension=embedding_dim, # Number of encoder features\n        module__dictionary_dimension=len(data_info.vocab.itos), # Dictionary dimension\n        module__output_dimension=3,\n        module__dropout=0.8,\n        module__pretrained=data_info.vocab.vectors,\n        module__continuous_features_dimension=train_features_continuous.shape[1],\n        module__categorical_feature_embedding_dimensions=[\n            (CATEGORICAL_FEATURES[c], 80) for c in CATEGORICAL_FEATURES\n        ],\n    )\n    model.fit(featurized_train_data, maybe_cuda(train_labels))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"d223052858198155a183745d929c4007fff50c99"},"cell_type":"code","source":"(FASTTEXT_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n FASTTEXT_TABULAR_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_for_fasttext_tabular_model,\n        train_fasttext_tabular_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"a078433228a239c01010e050ce49332cf521db4d"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        FASTTEXT_TABULAR_MODEL_TEST_PROBABILITIES,\n    ),\n    'fasttext_tabular_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ec6bc02de9a15e0ce4517ea35018d878ae89f06"},"cell_type":"markdown","source":"### 3.6.3 TextCNN\nHere we use a convolutional neural network on the text.\n\nWe kind of have to think of text like an \"image\". The horizontal\ndimension is just each sentence (of variable length, but they all have\npadding at the end). Then on the vertical axis, we have the word vectors\nand we move our window across. Idea here is that similar to images, the\norder between words and between word vector components probably matters,\nso we take into account conceptual space and sentence space in the text\nimage domain."},{"metadata":{"trusted":false,"_uuid":"36b5aa35a32675d50755594ee9bbf2f80bf22c88"},"cell_type":"code","source":"def featurize_dataframe_sequences_for_textcnn(*dataframes):\n    sequences_lists = tokenize_sequences(\n        *tuple(list(df['clean_description']) for df in dataframes)\n    )\n\n    text = torchtext_create_text_vocab(*sequences_lists,\n                                       vectors='glove.6B.100d')\n    \n    return text, tuple(text.process(sl).transpose(0, 1) for sl in sequences_lists)\n\ndef train_textcnn_model(data_info,\n                        featurized_train_data,\n                        featurized_validation_data,\n                        train_labels,\n                        validation_labels,\n                        train_param_grid_optimal=None):\n    embedding_dim = 100\n    model = NoToTensorInLossClassifier(\n        TextCNN,\n        lr=0.001,\n        batch_size=64,\n        optimizer=optim.Adam,\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='textcnn_checkpoint'),\n            TrainEndCheckpoint(dirname='textcnn_tabular_checkpoint',\n                               fn_prefix='textcnn_train_end_')\n        ],\n        max_epochs=10,\n        train_split=shuffled_train_test_split_by_indices(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        module__encoder_dimension=embedding_dim, # Number of encoder features\n        module__dictionary_dimension=len(data_info.vocab.itos), # Dictionary dimension\n        module__output_dimension=3,\n        module__n_filters=10,\n        module__filter_sizes=(3, 4, 5),\n        module__dropout=0.8,\n        module__pretrained=data_info.vocab.vectors\n    )\n    model.fit(featurized_train_data, maybe_cuda(train_labels))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"9f20ea788426c143596c3c67d8f9421183e450ab"},"cell_type":"code","source":"(TEXTCNN_MODEL_VALIDATION_PROBABILITIES,\n TEXTCNN_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_dataframe_sequences_for_textcnn,\n        train_textcnn_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"f95d22444f17b5b41c70c83db2cbd26e75b92e31"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        TEXTCNN_MODEL_TEST_PROBABILITIES,\n    ),\n    'textcnn_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ef43ece0f99add180dd8669f7807d0664a42f2c4"},"cell_type":"code","source":"def featurize_dataframes_for_textcnn_tabular_model(*dataframes):\n    data_info, model_datasets = featurize_dataframe_sequences_for_textcnn(*dataframes)\n    return data_info, splice_into_datasets(tuple([x] for x in model_datasets),\n                                           ((TRAIN_FEATURES_CONTINUOUS_TENSOR,\n                                             TRAIN_FEATURES_CATEGORICAL_TENSOR),\n                                            (VALIDATION_FEATURES_CONTINUOUS_TENSOR,\n                                             VALIDATION_FEATURES_CATEGORICAL_TENSOR),\n                                            (TEST_FEATURES_CONTINUOUS_TENSOR,\n                                             TEST_FEATURES_CATEGORICAL_TENSOR)))\n\ndef train_textcnn_tabular_model(data_info,\n                                featurized_train_data,\n                                featurized_validation_data,\n                                train_labels,\n                                validation_labels,\n                                train_param_grid_optimal=None):\n    embedding_dim = 100\n    _, train_features_continuous, _ = featurized_train_data\n    model = NoToTensorInLossClassifier(\n        TextCNNWithTabularData,\n        lr=0.001,\n        batch_size=256,\n        optimizer=optim.Adam,\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            LRAnnealing(),\n            LRScheduler(),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='fasttext_checkpoint'),\n            TrainEndCheckpoint(dirname='fasttext_tabular_checkpoint',\n                               fn_prefix='fasttext_train_end_')\n        ],\n        max_epochs=10,\n        train_split=shuffled_train_test_split_by_indices(0.3),\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        module__encoder_dimension=embedding_dim, # Number of encoder features\n        module__dictionary_dimension=len(data_info.vocab.itos), # Dictionary dimension\n        module__output_dimension=3,\n        module__n_filters=100,\n        module__filter_sizes=(3, 4, 5),\n        module__dropout=0.8,\n        module__pretrained=data_info.vocab.vectors,\n        module__continuous_features_dimension=train_features_continuous.shape[1],\n        module__categorical_feature_embedding_dimensions=[\n            (CATEGORICAL_FEATURES[c], 80) for c in CATEGORICAL_FEATURES\n        ],\n    )\n    model.fit(featurized_train_data, maybe_cuda(train_labels))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"6c82ec145c44ea70063ca8b3463be25dec5f7633"},"cell_type":"code","source":"(TEXTCNN_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n TEXTCNN_TABULAR_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_dataframes_for_textcnn_tabular_model,\n        train_textcnn_tabular_model,\n        predict_with_sklearn_estimator\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4a5e209f56ced3e816c5492b03f738d99b4f9601"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        TEXTCNN_TABULAR_MODEL_TEST_PROBABILITIES,\n    ),\n    'textcnn_tabular_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"970078e5b0f2fb8b4b296c94e4f4a00fe6778e71"},"cell_type":"markdown","source":"### 3.6.4 BERT\n\nBERT is a big ol' attentional model.\n\nWe use the pretrained version of BERT (eg `bert-base-uncased`) and the\ncorresponding tokenizer - `bert_featurize_dataframe` reads from\n`clean_description` in the dataframe and tokenizes sentences in the\nsame way that the pre-trained model was tokenized and should in principle\ndo the word vector mapping for us.\n\nEssentially what we are doing here is fine-tuning the classification\nlayers of BERT."},{"metadata":{"trusted":false,"_uuid":"88a337515a5f1561f0b43af1030125b67e364d81"},"cell_type":"code","source":"gc_and_clear_caches(None)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"0091b86e01fd2e6aefe6751287448f43538a6440"},"cell_type":"code","source":"def flatten_params(params):\n    return {\n        k: v[0] if isinstance(v, list) else v for k, v in params.items()\n    }","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"ce4335173150beed93bfcbe6d94703f608b2e4e3"},"cell_type":"code","source":"BERT_MODEL = 'bert-base-uncased'\n\n\ndef featurize_bert_lang_features(*dataframes):\n    return _, tuple(\n        tuple(torch.stack(x) for x in zip(*features))\n        for features in bert_featurize_data_frames(BERT_MODEL, *dataframes)\n    )\n        \n\ndef train_bert_lang_model(data_info,\n                          featurized_train_data,\n                          featurized_validation_data,\n                          train_labels,\n                          validation_labels,\n                          train_param_grid_optimal=None):\n    model = BertClassifier(\n        module=create_bert_model(BERT_MODEL, 3),\n        optimizer__warmup=WARMUP_PROPORTION,\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        optimizer=BertAdam,\n        lr=6e-5,\n        len_train_data=int(len(featurized_train_data[0])),\n        num_labels=3,\n        batch_size=16,\n        train_split=shuffled_train_test_split_by_indices(0.1),\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='bert_lang_checkpoint')\n        ],\n    )\n    \n    if not train_param_grid_optimal:\n        # As sugested by Maksad, need to do a hyperparameter search\n        # here to get good results.\n        param_grid = {\n            \"batch_size\": [16, 32],\n            \"lr\": [6e-5, 3e-5, 3e-1, 2e-5],\n            \"max_epochs\": [3, 4]\n        }\n        search = GridSearchCV(model,\n                              param_grid,\n                              cv=1,\n                              refit=False,\n                              scoring=make_scorer(log_loss,\n                                                  greater_is_better=False,\n                                                  needs_proba=True))\n        search.fit(TensorTuple(featurized_train_data), train_labels)\n\n        print('Best params {}'.format(search.best_params_))\n        # Now re-fit the estimator manually, using the best params -\n        # we do this manually since we need a different view over\n        # the training data to make it work\n        best = clone(search.estimator, safe=True).set_params(**search.best_params_)\n        best.fit(featurized_train_data, train_labels)\n        return best\n    else:\n        model = clone(model, safe=True).set_params(**flatten_params(train_param_grid_optimal))\n        model.fit(featurized_train_data, train_labels)\n        return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"5cee0621bbccae8d16bb53649124dc90ffe0283e"},"cell_type":"code","source":"(BERT_LANG_MODEL_VALIDATION_PROBABILITIES,\n BERT_LANG_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS,\n        VALIDATION_LABELS,\n        featurize_bert_lang_features,\n        train_bert_lang_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'lr': [2e-05],\n            'max_epochs': [4],\n            'batch_size': [32]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c9be672c0bf69db1e64432397dc6fc41c586e414"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        BERT_LANG_MODEL_TEST_PROBABILITIES,\n    ),\n    'bert_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"c1f8df057e8154d1f9a25126cb68895c7b13bd38"},"cell_type":"code","source":"def featurize_bert_tabular_features(*dataframes):\n    model_datasets = splice_into_datasets(tuple([x] for x in bert_featurize_data_frames(BERT_MODEL, *dataframes)),\n                                          ((TRAIN_FEATURES_CONTINUOUS_TENSOR,\n                                            TRAIN_FEATURES_CATEGORICAL_TENSOR),\n                                           (VALIDATION_FEATURES_CONTINUOUS_TENSOR,\n                                            VALIDATION_FEATURES_CATEGORICAL_TENSOR),\n                                           (TEST_FEATURES_CONTINUOUS_TENSOR,\n                                            TEST_FEATURES_CATEGORICAL_TENSOR)))\n    return _, tuple(\n        tuple(torch.stack(x) for x in zip(*[\n            tuple(list(bert_features) + [continuous, categorical])\n            for bert_features, continuous, categorical in zip(features,\n                                                              continuous_tensor,\n                                                              categorical_tensor)\n        ]))\n        for features, continuous_tensor, categorical_tensor in model_datasets\n    )\n\n\ndef train_bert_tabular_model(data_info,\n                             featurized_train_data,\n                             featurized_validation_data,\n                             train_labels,\n                             validation_labels,\n                             train_param_grid_optimal=None):\n    batch_size = 16\n    _, _, _, continuous_features, categorical_features = featurized_train_data\n    model = BertClassifier(\n        module=create_bert_model_with_tabular_features(\n            BERT_MODEL,\n            continuous_features.shape[1],\n            [\n                (CATEGORICAL_FEATURES[c], 80) for c in CATEGORICAL_FEATURES\n            ],\n            3\n        ),\n        len_train_data=int(len(featurized_train_data[0])),\n        optimizer__warmup=WARMUP_PROPORTION,\n        device='cuda' if torch.cuda.is_available() else 'cpu',\n        optimizer=BertAdam,\n        num_labels=3,\n        batch_size=batch_size,\n        train_split=shuffled_train_test_split_by_indices(0.3),\n        callbacks=[\n            SummarizeParameters(),\n            EpochScoring(scoring='accuracy'),\n            ProgressBar(),\n            CheckpointAndKeepBest(dirname='bert_lang_checkpoint')\n        ],\n    )\n    \n    if not train_param_grid_optimal:\n        # As sugested by Maksad, need to do a hyperparameter search\n        # here to get good results.\n        param_grid = {\n            \"batch_size\": [16, 32],\n            \"lr\": [6e-5, 3e-5, 3e-5, 2e-5],\n            \"max_epochs\": [3, 4]\n        }\n        search = GridSearchCV(model,\n                              param_grid,\n                              cv=2,\n                              refit=False,\n                              scoring=make_scorer(log_loss,\n                                                  greater_is_better=False,\n                                                  needs_proba=True))\n        search.fit(TensorTuple(featurized_train_data), train_labels)\n\n        print('Best params {}'.format(search.best_params_))\n        # Now re-fit the estimator manually, using the best params -\n        # we do this manually since we need a different view over\n        # the training data to make it work\n        best = clone(search.estimator, safe=True).set_params(**search.best_params_)\n        best.fit(featurized_train_data, train_labels)\n        return best\n    else:\n        model = clone(model, safe=True).set_params(**flatten_params(train_param_grid_optimal))\n        model.fit(featurized_train_data, train_labels)\n        return model","execution_count":null,"outputs":[]},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"cea5bb4a4f2fff09a47679b6b2fbc14d1dddc240"},"cell_type":"code","source":"(BERT_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n BERT_TABULAR_MODEL_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        TRAIN_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TEST_DATAFRAME,\n        VALIDATION_DATAFRAME,\n        TRAIN_LABELS_TENSOR,\n        VALIDATION_LABELS_TENSOR,\n        featurize_bert_tabular_features,\n        train_bert_tabular_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'lr': [2e-05],\n            'max_epochs': [4],\n            'batch_size': [32]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"d510c6f3e58e9cb13592ba7aa121d8667fb64977"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        BERT_TABULAR_MODEL_TEST_PROBABILITIES,\n    ),\n    'bert_tabular_model_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2c34aa94e622c306d5268387a2956db21cb6fbbd"},"cell_type":"markdown","source":"### 3.6.5 ULMFiT\nThis is basically transfer learning onto an AWD-LSTM (used by fastai).\n\nWe have to use the fastai API here directly, since the best\nimplementation I found of this was by Jeremy Howard himself.\n\nNote before we ran the notebook, we used `train_lm.py` (pointing it\nto the training data) to fine-tune the existing language model,\nwhich in principle is good at English (WikiText-103) and made transferred\nwhat we knew about English word prediction to predicting RentHop\ndescriptions, fine tuning on the RentHop description task. This is\ndifferent from just using word vectors, since you get the benefit\nof the entire language model and contextual information, not just\nthe vectors themselves.\n\nIt is critical here that we load the same vocabulary used to fine\ntune the network into the model when we load in the weights. Also,\nwe need to do the train-test split ourselves, because there\nis a bug in the library where the vocabulary is only computed\non the training set and not the validation set, meaning that if the\nsplits are random you could miss words.\n\nUnfortunately, the fastai API is very involved, making it difficult\nto wrap with skortch without breaking stuff, so we have to use it\nin a slightly different way to do the same thing. We can also only\nreport statistics on the first batch, hopefully that should be enough."},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"aec4f3aa59f853d7a0e3d9c473f7af68dde8d313"},"cell_type":"code","source":"from utils.language_models.ulmfit import train_ulmfit_model_and_get_validation_and_test_set_predictions\n\n(ULMFIT_VALIDATION_PROBABILITIES,\n ULMFIT_TEST_PROBABILITIES) = train_ulmfit_model_and_get_validation_and_test_set_predictions(\n    TRAIN_DATAFRAME,\n    VALIDATION_DATAFRAME,\n    TEST_DATAFRAME\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"83c4df26e5f34405e62ac8d25823cb4a0f8caea8"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        ULMFIT_TEST_PROBABILITIES\n    ),\n    'ulmfit_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"573b8d8b157d70770e8b5936996f541a83f393c1"},"cell_type":"markdown","source":"### 4 Ensembling\n\nNow that we have all of our models and various predictions, we can\nensemble them together in the form of one big logistic regression model or gradient boosted tree to work out what the \"true\" classes are based on how all the different models were voting.\n\nNotice that the confusion matrix for each of the models was quite\ndifferent - this indicates that each model is probably more biased towards certain features. We can get the predictions for all of our training data and put them together into another dataset which we then apply XGBoost and a linear model to. In principle this allows us to decide amongst the learners by comparing their probability distributions.\n\nWe also only do the stacking on the validation set by splitting the validation set again into training and validation data - we don't use the  original training data since that's what the learners themselves were trained on."},{"metadata":{"trusted":false,"_uuid":"659d7dec323bdec380497219b9b7227bceaf396a"},"cell_type":"code","source":"(STACKED_VALIDATION_PREDICTIONS_TRAINING_SET,\n STACKED_VALIDATION_PREDICTIONS_VALIDATION_SET,\n _,\n VALIDATION_SPLIT_VALIDATION_DATAFRAME,\n TRAINING_SPLIT_FEATURES_VALIDATION_DATAFRAME,\n VALIDATION_SPLIT_FEATURES_VALIDATION_DATAFRAME,\n STACKED_VALIDATION_PREDICTIONS_LABELS_TRAINING_SET,\n STACKED_VALIDATION_PREDICTIONS_LABELS_VALIDATION_SET) = train_test_split(\n    np.column_stack([\n        LOGISTIC_REGRESSION_MODEL_VALIDATION_PROBABILITIES,\n        XGBOOST_MODEL_VALIDATION_PROBABILITIES,\n        RF_MODEL_VALIDATION_PROBABILITIES,\n        ADABOOST_MODEL_VALIDATION_PROBABILITIES,\n        EXTRATREES_MODEL_VALIDATION_PROBABILITIES,\n        SVC_MODEL_VALIDATION_PROBABILITIES,\n        SIMPLE_RNN_MODEL_VALIDATION_PROBABILITIES,\n        SIMPLE_RNN_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n        FASTTEXT_MODEL_VALIDATION_PROBABILITIES,\n        FASTTEXT_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n        TEXTCNN_MODEL_VALIDATION_PROBABILITIES,\n        TEXTCNN_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n        BERT_LANG_MODEL_VALIDATION_PROBABILITIES,\n        BERT_TABULAR_MODEL_VALIDATION_PROBABILITIES,\n        ULMFIT_VALIDATION_PROBABILITIES,\n    ]),\n    VALIDATION_DATAFRAME,\n    FEATURES_VALIDATION_DATAFRAME,\n    VALIDATION_LABELS,\n    stratify=VALIDATION_LABELS,\n    test_size=0.1\n)\n\nSTACKED_TEST_PREDICTIONS_TEST_SET = np.column_stack([\n    LOGISTIC_REGRESSION_MODEL_TEST_PROBABILITIES,\n    XGBOOST_MODEL_TEST_PROBABILITIES,\n    RF_MODEL_TEST_PROBABILITIES,\n    ADABOOST_MODEL_TEST_PROBABILITIES,\n    EXTRATREES_MODEL_TEST_PROBABILITIES,\n    SVC_MODEL_TEST_PROBABILITIES,\n    SIMPLE_RNN_MODEL_TEST_PROBABILITIES,\n    SIMPLE_RNN_TABULAR_MODEL_TEST_PROBABILITIES,\n    FASTTEXT_MODEL_TEST_PROBABILITIES,\n    FASTTEXT_TABULAR_MODEL_TEST_PROBABILITIES,\n    TEXTCNN_MODEL_TEST_PROBABILITIES,\n    TEXTCNN_TABULAR_MODEL_TEST_PROBABILITIES,\n    BERT_LANG_MODEL_TEST_PROBABILITIES,\n    BERT_TABULAR_MODEL_TEST_PROBABILITIES,\n    ULMFIT_TEST_PROBABILITIES,\n])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"591301f14067b892d2b2cfe7abf972129875a579"},"cell_type":"markdown","source":"### 4.1 Ensembling with XGBoost"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"d41bcca0d4f5d662bfb280f52b50cd2d59cb07e3"},"cell_type":"code","source":"def identity_unpack(*args):\n    return _, args\n\n(XGBOOST_MODEL_STACKED_VALIDATION_PROBABILITIES,\n XGBOOST_MODEL_STACKED_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_TRAINING_SET),\n        pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_VALIDATION_SET),\n        pd.DataFrame(STACKED_TEST_PREDICTIONS_TEST_SET),\n        VALIDATION_SPLIT_VALIDATION_DATAFRAME,\n        STACKED_VALIDATION_PREDICTIONS_LABELS_TRAINING_SET.reset_index(drop=True),\n        STACKED_VALIDATION_PREDICTIONS_LABELS_VALIDATION_SET.reset_index(drop=True),\n        identity_unpack,\n        train_xgboost_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'colsample_bytree': [0.8],\n            'gamma': [2],\n            'max_depth': [3],\n            'min_child_weight': [5],\n            'n_estimators': [100],\n            'subsample': [0.8]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"2c008ba7d1249a2015c9cc46ed782a518abf2fdf"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        XGBOOST_MODEL_STACKED_TEST_PROBABILITIES\n    ),\n    'xgboost_stacked_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"30d8215471e317ef78c58fc1fd1b7711ec6f0136"},"cell_type":"markdown","source":"### 4.2 Ensembling with Logistic Regression"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"48a6e354b44ce646dcba7d0b9371e96f888009cf"},"cell_type":"code","source":"(LOGISTIC_REGRESSION_MODEL_STACKED_VALIDATION_PROBABILITIES,\n LOGISTIC_REGRESSION_MODEL_STACKED_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_TRAINING_SET),\n        pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_VALIDATION_SET),\n        pd.DataFrame(STACKED_TEST_PREDICTIONS_TEST_SET),\n        VALIDATION_SPLIT_VALIDATION_DATAFRAME,\n        STACKED_VALIDATION_PREDICTIONS_LABELS_TRAINING_SET.reset_index(drop=True),\n        STACKED_VALIDATION_PREDICTIONS_LABELS_VALIDATION_SET.reset_index(drop=True),\n        identity_unpack,\n        train_logistic_regression_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'C': [1.0],\n            'class_weight': [None],\n            'penalty': ['l2']\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"92676a4bafd90874ea9e945d0abd599699fa430d"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        LOGISTIC_REGRESSION_MODEL_STACKED_TEST_PROBABILITIES\n    ),\n    'logistic_stacked_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7efe825dfcf7a6dda865d1900304e059ee7b4e37"},"cell_type":"markdown","source":"### 4.3 Guided Ensembles\n\nWe can also try to \"guide\" the ensembles by concatenating our features and then doing predictions based on that.\n\nUnfortunately this seems to be more of a distraction than a help - we actually perform *worse* on validation set once we start introducing our original data back in."},{"metadata":{"_uuid":"291433b2ba2f8883b96badfaa632ea23693add7e"},"cell_type":"markdown","source":"#### 4.3.1 Guided Ensemble - XGBoost"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"0acee429e03004d00b738c032a673307de2049d9"},"cell_type":"code","source":"(GUIDED_XGBOOST_MODEL_STACKED_VALIDATION_PROBABILITIES,\n GUIDED_XGBOOST_MODEL_STACKED_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        pd.concat((pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_TRAINING_SET),\n                   TRAINING_SPLIT_FEATURES_VALIDATION_DATAFRAME.reset_index().drop([\"index\"], axis=1)), axis=1),\n        pd.concat((pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_VALIDATION_SET),\n                   VALIDATION_SPLIT_FEATURES_VALIDATION_DATAFRAME.reset_index().drop([\"index\"], axis=1)), axis=1),\n        pd.concat((pd.DataFrame(STACKED_TEST_PREDICTIONS_TEST_SET),\n                   FEATURES_TEST_DATAFRAME.reset_index().drop(\"index\", axis=1)), axis=1),\n        VALIDATION_SPLIT_VALIDATION_DATAFRAME,\n        STACKED_VALIDATION_PREDICTIONS_LABELS_TRAINING_SET.reset_index(drop=True),\n        STACKED_VALIDATION_PREDICTIONS_LABELS_VALIDATION_SET.reset_index(drop=True),\n        featurize_for_tree_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_xgboost_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'colsample_bytree': [0.6],\n            'gamma': [2],\n            'max_depth': [3],\n            'min_child_weight': [1],\n            'n_estimators': [100],\n            'subsample': [0.8]\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"55a7acef30e4ddbd9da1fa5370a35cd483c0688f"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        GUIDED_XGBOOST_MODEL_STACKED_TEST_PROBABILITIES\n    ),\n    'guided_xgboost_stacked_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"885abbfeca50dd5354a1a61b9ef9a782ad809f68"},"cell_type":"markdown","source":"#### 4.3.2 Guided Ensemble - Logistic Regression"},{"metadata":{"scrolled":false,"trusted":false,"_uuid":"90edcd0c8124d8de7b21ee4c0be7273b3f9dc6fd"},"cell_type":"code","source":"(GUIDED_LOGISTIC_MODEL_STACKED_VALIDATION_PROBABILITIES,\n GUIDED_LOGISTIC_MODEL_STACKED_TEST_PROBABILITIES) = gc_and_clear_caches(\n    train_model_and_get_validation_and_test_set_predictions(\n        pd.concat((pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_TRAINING_SET),\n                   TRAINING_SPLIT_FEATURES_VALIDATION_DATAFRAME.reset_index().drop([\"index\"], axis=1)), axis=1),\n        pd.concat((pd.DataFrame(STACKED_VALIDATION_PREDICTIONS_VALIDATION_SET),\n                   VALIDATION_SPLIT_FEATURES_VALIDATION_DATAFRAME.reset_index().drop([\"index\"], axis=1)), axis=1),\n        pd.concat((pd.DataFrame(STACKED_TEST_PREDICTIONS_TEST_SET),\n                   FEATURES_TEST_DATAFRAME.reset_index().drop(\"index\", axis=1)), axis=1),\n        VALIDATION_SPLIT_VALIDATION_DATAFRAME,\n        STACKED_VALIDATION_PREDICTIONS_LABELS_TRAINING_SET.reset_index(drop=True),\n        STACKED_VALIDATION_PREDICTIONS_LABELS_VALIDATION_SET.reset_index(drop=True),\n        featurize_for_tabular_models(DROP_COLUMNS, CATEGORICAL_FEATURES),\n        train_logistic_regression_model,\n        predict_with_sklearn_estimator,\n        train_param_grid_optimal={\n            'C': [1.0],\n            'class_weight': [None],\n            'penalty': ['l2']\n        }\n    )\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":false,"_uuid":"4d2498406b9a3e78cefdfb59329ff20b9965c817"},"cell_type":"code","source":"write_predictions_table_to_csv(\n    get_prediction_probabilities_with_columns_from_predictions(\n        FEATURES_TEST_DATAFRAME['listing_id'],\n        GUIDED_LOGISTIC_MODEL_STACKED_TEST_PROBABILITIES\n    ),\n    'guided_logistic_stacked_submissions.csv'\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e856f45d6f6f7467c7ad41c9baef651359ad4f14"},"cell_type":"markdown","source":"## 5 Conclusions\n\nWith all the work that we did, we were able to reduce the validation loss from a baseline linear model $\\approx 0.63$ to a much better $\\approx 0.537$.\n\nOn the Kaggle I scored $\\approx 0.58836$. Not great, but also not too bad either. Possibly other generated submissions may have scored better.\n\nEnsembling provided us the biggest benefit - we were able to combine all the knowledge we gained from different learners over different parts of the training data to make better classifications over the validation data.\n\nI was able to apply quite a lot of the knowledge I gained from this course into forming my model. In particular:\n * Where to look for guidance on the competition (eg, the forums, winning kernels, etc)\n * A better exploratory data analysis using Seaborn crossplots to look at correlation between the numerical features.\n * Feature engineering on numerical data by performing simple math, this helps decision tree based models that can't do this kind of transformation inherently.\n * Correct usage of sklearn - even though I wasn't able to use Pipelines very effectively, I was able to make use of meta-estimators like grid-search and leverage the estimator API.\n * Usage of ensembling methods - as stated before, the use of ensembling over may different methods provided quite a large benefit. I am sure that this benefit would have been even greater if we used bagging or boosting on the actual estimators themselves with subsets of the training data, though I didn't have enough time to implement that properly.\n * Through my own research, I learned quite a lot about language models and text classification approaches (FastText, TextCNN, BERT)."}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.8"}},"nbformat":4,"nbformat_minor":1}