{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import MinMaxScaler, LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nimport numpy as np\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.model_selection import StratifiedKFold\nimport catboost\nfrom sklearn.metrics import log_loss\nfrom sklearn.linear_model import LogisticRegression\nimport warnings\nimport zipfile  \n\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-18T14:43:44.319373Z","iopub.execute_input":"2022-02-18T14:43:44.319757Z","iopub.status.idle":"2022-02-18T14:43:46.040206Z","shell.execute_reply.started":"2022-02-18T14:43:44.319656Z","shell.execute_reply":"2022-02-18T14:43:46.039189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = None  \ndata = None  \nwith zipfile.ZipFile(\"../input/two-sigma-connect-rental-listing-inquiries/train.json.zip\", \"r\") as z:\n    for filename in z.namelist():  \n        print(filename)  \n        with z.open(filename) as f:  \n            data = f.read()  \n            d = json.loads(data.decode(\"utf-8\")) ","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:43:56.404807Z","iopub.execute_input":"2022-02-18T14:43:56.405573Z","iopub.status.idle":"2022-02-18T14:43:58.790506Z","shell.execute_reply.started":"2022-02-18T14:43:56.405525Z","shell.execute_reply":"2022-02-18T14:43:58.789738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert into df\ntrain = pd.DataFrame.from_dict(d)\ntrain.reset_index(level=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:44:10.632340Z","iopub.execute_input":"2022-02-18T14:44:10.632685Z","iopub.status.idle":"2022-02-18T14:44:10.941709Z","shell.execute_reply.started":"2022-02-18T14:44:10.632647Z","shell.execute_reply":"2022-02-18T14:44:10.940868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:44:16.419999Z","iopub.execute_input":"2022-02-18T14:44:16.420560Z","iopub.status.idle":"2022-02-18T14:44:16.452351Z","shell.execute_reply.started":"2022-02-18T14:44:16.420523Z","shell.execute_reply":"2022-02-18T14:44:16.451691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Preprocessing","metadata":{}},{"cell_type":"code","source":"# replace empty rows with NaN\ntrain = train.replace(r'^\\s*$', np.nan, regex=True)\nle = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:44:55.322857Z","iopub.execute_input":"2022-02-18T14:44:55.323535Z","iopub.status.idle":"2022-02-18T14:44:56.631960Z","shell.execute_reply.started":"2022-02-18T14:44:55.323490Z","shell.execute_reply":"2022-02-18T14:44:56.630972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Preprocessing:\n    def __init__(self, data, le):\n        self.data = data\n        self.le = le\n        \n    def feature_eng(self) -> pd.DataFrame:\n        ###### Column description\n        # if text exists then replace by 1, otherwise by 0\n        self.data['description'][self.data['description'].isna()] = 0\n        self.data['description'][self.data['description']!=0] = 1\n        ###### Column features\n        # extract text from list\n        self.data['features'] = self.data['features'].astype(str)\n        self.data['features'] = [','.join(map(str, i)) for i in self.data['features']]\n        # remove punctuation marks and spaces\n        self.data['features'] = self.data['features'].str.replace(r'[^\\w\\s]+', '')\n        self.data['features'] = self.data['features'].str.replace(' ', '')\n        ###### Column created\n        # extract week day from created column - create weekday column\n        # delete created column\n        self.data['created'] =  pd.to_datetime(self.data['created'], format='%Y-%m-%d %H:%M:%S')\n        self.data['weekday'] = self.data['created'].dt.dayofweek\n        self.data.drop('created', axis=1, inplace=True)\n        ###### Column street_address\n        # remove spaces\n        self.data['street_address'] = self.data['street_address'].astype(str)\n        self.data['street_address'] = self.data['street_address'].str.replace(' ', '')\n\n        return self.data\n    \n    def tranform_feat(self) -> pd.DataFrame:\n        '''\n        Encode categorical features, scale\n        numerical features and expand in pca\n        '''\n        data = self.feature_eng()\n        # labelencoding to 'features', 'street_address' and 'manager_id'\n        data['features'] = self.le.fit_transform(data['features'].to_list())\n        data['manager_id'] = data['manager_id'].astype(str)\n        data['manager_id'] = self.le.fit_transform(data['manager_id'].to_list())\n        data['street_address'] = self.le.fit_transform(data['street_address'].to_list())\n        # minmaxscaler and pca for numeric features\n        numeric_features = data._get_numeric_data().columns\n        scaler = MinMaxScaler()\n        pca = PCA()\n        for nf in numeric_features:\n            data[nf] = scaler.fit_transform(pd.DataFrame(data[nf]))\n            data[nf] = pca.fit_transform(pd.DataFrame(data[nf]))\n\n        return data","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:48:27.553757Z","iopub.execute_input":"2022-02-18T14:48:27.554071Z","iopub.status.idle":"2022-02-18T14:48:27.570674Z","shell.execute_reply.started":"2022-02-18T14:48:27.554038Z","shell.execute_reply":"2022-02-18T14:48:27.569701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transform train data\ntransformed_train = Preprocessing(train, le).tranform_feat()\ntransformed_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:48:52.325208Z","iopub.execute_input":"2022-02-18T14:48:52.325494Z","iopub.status.idle":"2022-02-18T14:48:55.550766Z","shell.execute_reply.started":"2022-02-18T14:48:52.325462Z","shell.execute_reply":"2022-02-18T14:48:55.550126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete target and others columns\nX = transformed_train.drop(labels=['index', 'building_id', 'listing_id', 'photos', 'interest_level',\n                      'display_address'], axis=1)\n# stay only target \ny = transformed_train['interest_level']","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:49:27.088411Z","iopub.execute_input":"2022-02-18T14:49:27.089218Z","iopub.status.idle":"2022-02-18T14:49:27.101502Z","shell.execute_reply.started":"2022-02-18T14:49:27.089171Z","shell.execute_reply":"2022-02-18T14:49:27.100612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I stayed the features: bathrooms, bedrooms, description (in the form of a categorical feature - there is or is not a description), features, latitude, longitude, price, street_address и weekday","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:50:22.218661Z","iopub.execute_input":"2022-02-18T14:50:22.219339Z","iopub.status.idle":"2022-02-18T14:50:22.239775Z","shell.execute_reply.started":"2022-02-18T14:50:22.219290Z","shell.execute_reply":"2022-02-18T14:50:22.238679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"__Building a multiclass classification model__\n1. Search parameters with RandomizedSearchCV\n2. Implementation of cross-validation with splitting into three parts\n3. Building RandomForestClassifier, CatBoostClassifier и LogisticRegression\n4. Estimating models with logloss","metadata":{}},{"cell_type":"code","source":"def rand_search(params, model):\n    '''cross-validation and randomizedsearch'''\n    folds = 3\n    param_comb = 5\n    skf = StratifiedKFold(n_splits=folds, shuffle=True)\n    random_search = RandomizedSearchCV(model, param_distributions=params,\n                                   n_iter=param_comb, scoring='neg_log_loss', \n                                   n_jobs=4, cv=skf.split(X_train, y_train),\n                                   verbose=0)\n    return random_search","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:52:01.076228Z","iopub.execute_input":"2022-02-18T14:52:01.076717Z","iopub.status.idle":"2022-02-18T14:52:01.083789Z","shell.execute_reply.started":"2022-02-18T14:52:01.076670Z","shell.execute_reply":"2022-02-18T14:52:01.082721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# search parameters for RandomForestClassifier\nparams_rf = {\n        'n_estimators': [100, 500, 1000],\n        'min_samples_split': [2, 3, 4],\n        'max_depth': [3, 4, 5, 10]\n        }\nrf = RandomForestClassifier()\nrandom_search_rf = rand_search(params_rf, rf)\nrandom_search_rf.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:52:23.146129Z","iopub.execute_input":"2022-02-18T14:52:23.146897Z","iopub.status.idle":"2022-02-18T14:53:16.934369Z","shell.execute_reply.started":"2022-02-18T14:52:23.146826Z","shell.execute_reply":"2022-02-18T14:53:16.933299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# search parameters for LogisticRegression\nparams_lr = {\n        'solver': ['newton-cg', 'lbfgs', 'liblinear', 'sag', 'saga']\n        }\nlr = LogisticRegression()\nrandom_search_lr = rand_search(params_lr, lr)\nrandom_search_lr.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:53:16.936782Z","iopub.execute_input":"2022-02-18T14:53:16.937603Z","iopub.status.idle":"2022-02-18T14:53:23.848001Z","shell.execute_reply.started":"2022-02-18T14:53:16.937549Z","shell.execute_reply":"2022-02-18T14:53:23.846950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('The best parameters for RandomForestClassifier')\nprint(random_search_rf.best_params_)\nprint('The best parameters for LogisticRegression')\nprint(random_search_lr.best_params_)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:53:23.849653Z","iopub.execute_input":"2022-02-18T14:53:23.850149Z","iopub.status.idle":"2022-02-18T14:53:23.864533Z","shell.execute_reply.started":"2022-02-18T14:53:23.850104Z","shell.execute_reply":"2022-02-18T14:53:23.863089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_best = RandomForestClassifier(n_estimators=500, \n                                 min_samples_split=4,\n                                 max_depth=4)\n\nrf_best.fit(X_train, y_train)\npred_rf = rf_best.predict_proba(X_test)\nprint('logloss RandomForestClassifier =', log_loss(y_test, pred_rf))","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:54:03.039272Z","iopub.execute_input":"2022-02-18T14:54:03.040296Z","iopub.status.idle":"2022-02-18T14:54:15.294588Z","shell.execute_reply.started":"2022-02-18T14:54:03.040238Z","shell.execute_reply":"2022-02-18T14:54:15.293466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr_best = LogisticRegression(solver='newton-cg')\n\nlr_best.fit(X_train, y_train)\npred_lr = lr_best.predict_proba(X_test)\nprint('logloss LogisticRegression =', log_loss(y_test, pred_lr))","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:54:15.296198Z","iopub.execute_input":"2022-02-18T14:54:15.296558Z","iopub.status.idle":"2022-02-18T14:54:17.836693Z","shell.execute_reply.started":"2022-02-18T14:54:15.296516Z","shell.execute_reply":"2022-02-18T14:54:17.835756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# standard Catboost model\nct_best = catboost.CatBoostClassifier(loss_function='MultiClass')\n\nct_best.fit(X_train, y_train)\npred_ct = ct_best.predict_proba(X_test)\nprint('logloss CatBoostClassifier =', log_loss(y_test, pred_ct))","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:54:42.594396Z","iopub.execute_input":"2022-02-18T14:54:42.594709Z","iopub.status.idle":"2022-02-18T14:54:54.758982Z","shell.execute_reply.started":"2022-02-18T14:54:42.594677Z","shell.execute_reply":"2022-02-18T14:54:54.758242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using the standard catboost, we managed to get the smallest logloss = 0.6.","metadata":{}},{"cell_type":"code","source":"# create submission\nzf = zipfile.ZipFile('../input/two-sigma-connect-rental-listing-inquiries/sample_submission.csv.zip') \nsample_submission = pd.read_csv(zf.open('sample_submission.csv'))","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:57:56.422524Z","iopub.execute_input":"2022-02-18T14:57:56.422959Z","iopub.status.idle":"2022-02-18T14:57:56.530147Z","shell.execute_reply.started":"2022-02-18T14:57:56.422916Z","shell.execute_reply":"2022-02-18T14:57:56.528576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d = None  \ndata = None  \nwith zipfile.ZipFile(\"../input/two-sigma-connect-rental-listing-inquiries/test.json.zip\", \"r\") as z:\n    for filename in z.namelist():  \n        print(filename)  \n        with z.open(filename) as f:  \n            data = f.read()  \n            d = json.loads(data.decode(\"utf-8\")) \n# convert to df\ntest = pd.DataFrame.from_dict(d)\ntest.reset_index(level=0, inplace=True)\n\ntransformed_test = Preprocessing(test, le).tranform_feat()\nX = transformed_test.drop(labels=['index', 'building_id', 'listing_id', 'photos',\n                      'display_address'], axis=1)\n\npred_test = ct_best.predict_proba(X)\npred_test_df = pd.DataFrame({'high':pred_test[:, 0], 'medium':pred_test[:, 1],\t'low':pred_test[:, 2]})\npred_test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:58:18.925302Z","iopub.execute_input":"2022-02-18T14:58:18.925610Z","iopub.status.idle":"2022-02-18T14:58:31.483354Z","shell.execute_reply.started":"2022-02-18T14:58:18.925579Z","shell.execute_reply":"2022-02-18T14:58:31.478836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission.drop(columns=['high', 'medium', 'low'], inplace=True)\nresult = pd.concat([sample_submission, pred_test_df], axis=1)\nresult.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:58:31.492191Z","iopub.execute_input":"2022-02-18T14:58:31.494109Z","iopub.status.idle":"2022-02-18T14:58:31.554606Z","shell.execute_reply.started":"2022-02-18T14:58:31.493945Z","shell.execute_reply":"2022-02-18T14:58:31.549930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result.to_csv('submission_two_sigma.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-18T14:58:37.941182Z","iopub.execute_input":"2022-02-18T14:58:37.941926Z","iopub.status.idle":"2022-02-18T14:58:38.546473Z","shell.execute_reply.started":"2022-02-18T14:58:37.941886Z","shell.execute_reply":"2022-02-18T14:58:38.545564Z"},"trusted":true},"execution_count":null,"outputs":[]}]}