{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.feature_extraction.text import TfidfVectorizer, CountVectorizer\nfrom sklearn.linear_model import Ridge, LinearRegression\nfrom sklearn.pipeline import Pipeline, FeatureUnion\nfrom sklearn.base import TransformerMixin, BaseEstimator\n\nimport re \nimport scipy\nfrom scipy import sparse\nimport gc \n\nfrom IPython.display import display\nfrom pprint import pprint\nfrom matplotlib import pyplot as plt \nfrom tqdm import tqdm \nimport time\nimport scipy.optimize as optimize\nimport warnings\nwarnings.filterwarnings(\"ignore\")\npd.options.display.max_colwidth=300\npd.options.display.max_columns = 100\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T22:50:32.904814Z","iopub.execute_input":"2022-02-05T22:50:32.905373Z","iopub.status.idle":"2022-02-05T22:50:32.913594Z","shell.execute_reply.started":"2022-02-05T22:50:32.905334Z","shell.execute_reply":"2022-02-05T22:50:32.912551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T22:50:33.359294Z","iopub.execute_input":"2022-02-05T22:50:33.359553Z","iopub.status.idle":"2022-02-05T22:50:33.372065Z","shell.execute_reply.started":"2022-02-05T22:50:33.359523Z","shell.execute_reply":"2022-02-05T22:50:33.371378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/comments_to_score.csv\")\nif len(test) < 10000:\n    df = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/validation_data.csv\").head(10)\n    test = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/comments_to_score.csv\").head(10)\nelse:\n    df = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/validation_data.csv\")\n    test = pd.read_csv(\"/kaggle/input/jigsaw-toxic-severity-rating/comments_to_score.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T22:50:33.789173Z","iopub.execute_input":"2022-02-05T22:50:33.78944Z","iopub.status.idle":"2022-02-05T22:50:34.072313Z","shell.execute_reply.started":"2022-02-05T22:50:33.78941Z","shell.execute_reply":"2022-02-05T22:50:34.071556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# validation data to training data \ndef valid_to_regression_df(df):\n    less_toxic_count = df[\"less_toxic\"].value_counts()\n    more_toxic_count = df[\"more_toxic\"].value_counts()\n    \n    all_toxic = pd.DataFrame(less_toxic_count).join(pd.DataFrame(more_toxic_count),how = \"outer\")\n    all_toxic = all_toxic.reset_index() \n    all_toxic = all_toxic.rename({\"index\":\"text\"},axis = 1)\n    all_toxic = all_toxic.fillna(0)\n    \n    all_toxic[\"sum\"] = all_toxic[\"less_toxic\"] + all_toxic[\"more_toxic\"]\n    all_toxic[\"total_score\"] = all_toxic[\"more_toxic\"] / all_toxic[\"sum\"]\n    all_toxic = all_toxic.drop([\"sum\",\"less_toxic\",\"more_toxic\"],axis = 1)\n    return all_toxic\ndf = valid_to_regression_df(df)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T22:50:41.388144Z","iopub.execute_input":"2022-02-05T22:50:41.388713Z","iopub.status.idle":"2022-02-05T22:50:41.409202Z","shell.execute_reply.started":"2022-02-05T22:50:41.388648Z","shell.execute_reply":"2022-02-05T22:50:41.408563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# make feature ","metadata":{}},{"cell_type":"markdown","source":"## Bert Features","metadata":{}},{"cell_type":"code","source":"## install detoxify from dataset\n!cp -r ../input/detoxify-sourcemodels/detoxify .\n!pip install -q ./detoxify\n!rm -r ./detoxify\n\n\n## copy detoxify pretrained models and transformers configuration files from dataset to local caches\n!mkdir -p  /root/.cache/torch/hub/checkpoints\n!mkdir -p  /root/.cache/huggingface/transformers\n!cp -r ../input/detoxify-sourcemodels/torch/hub/checkpoints /root/.cache/torch/hub\n!cp -r ../input/detoxify-sourcemodels/huggingface/transformers /root/.cache/huggingface\n\n\n# Setting environment variable TRANSFORMERS_OFFLINE=1 will tell Transformers to use local files only and will not try to look things up.\n# It’s possible to run Transformers in a firewalled or a no-network environment or in a Kaggle inference kernel !\nimport os\nos.environ[\"TRANSFORMERS_OFFLINE\"] = \"1\"","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:41:56.27835Z","iopub.execute_input":"2022-02-05T21:41:56.278792Z","iopub.status.idle":"2022-02-05T21:42:54.998898Z","shell.execute_reply.started":"2022-02-05T21:41:56.278753Z","shell.execute_reply":"2022-02-05T21:42:54.997982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from detoxify import Detoxify\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:55.001068Z","iopub.execute_input":"2022-02-05T21:42:55.001347Z","iopub.status.idle":"2022-02-05T21:42:56.484979Z","shell.execute_reply.started":"2022-02-05T21:42:55.001309Z","shell.execute_reply":"2022-02-05T21:42:56.48425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"huggingface_config_path = '../input/bert-base-uncased'","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:56.486241Z","iopub.execute_input":"2022-02-05T21:42:56.486498Z","iopub.status.idle":"2022-02-05T21:42:56.491494Z","shell.execute_reply.started":"2022-02-05T21:42:56.486464Z","shell.execute_reply":"2022-02-05T21:42:56.490824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_max_length=512","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:56.493664Z","iopub.execute_input":"2022-02-05T21:42:56.494287Z","iopub.status.idle":"2022-02-05T21:42:56.499718Z","shell.execute_reply.started":"2022-02-05T21:42:56.49425Z","shell.execute_reply":"2022-02-05T21:42:56.499024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detoxify\ndef f1 (df,target_col = \"text\"):\n    d =  Detoxify( model_type='original' ,device='cuda')\n\n    uniqeu_text_list = df[target_col].unique()\n    uniqeu_text_list = list(uniqeu_text_list)\n    \n    toxic_list = []\n    for text in tqdm(uniqeu_text_list):\n        result = d.predict(text[:512])\n        toxic_list.append(result)\n        \n    temp = pd.DataFrame()\n    temp[target_col] = uniqeu_text_list\n    temp[\"di\"] = toxic_list\n    cols = toxic_list[0].keys()\n    \n    for col in cols:\n        temp[col] = temp[\"di\"].apply(lambda x:x[col])\n    temp = temp.drop([\"di\"],axis = 1)\n    df = pd.merge(df,temp,on = target_col,suffixes = [\"\",\"_original\"])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:56.500718Z","iopub.execute_input":"2022-02-05T21:42:56.501409Z","iopub.status.idle":"2022-02-05T21:42:56.510433Z","shell.execute_reply.started":"2022-02-05T21:42:56.501372Z","shell.execute_reply":"2022-02-05T21:42:56.509704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detoxify _unbiased\ndef f2 (df,target_col = \"text\"):\n    huggingface_config_path = '../input/bert-base-uncased'\n    d =  Detoxify('unbiased', device='cuda')\n\n    uniqeu_text_list = df[target_col].unique()\n    uniqeu_text_list = list(uniqeu_text_list)\n    \n    toxic_list = []\n    for text in tqdm(uniqeu_text_list):\n        result = d.predict(text[:512])\n        toxic_list.append(result)\n        \n    temp = pd.DataFrame()\n    temp[target_col] = uniqeu_text_list\n    temp[\"di\"] = toxic_list\n    cols = toxic_list[0].keys()\n    \n    for col in cols:\n        temp[col + \"_unbiased\"] = temp[\"di\"].apply(lambda x:x[col])\n    temp = temp.drop([\"di\"],axis = 1)\n    df = pd.merge(df,temp,on = target_col,suffixes = [\"\",\"_unbiased\"])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:56.513253Z","iopub.execute_input":"2022-02-05T21:42:56.513564Z","iopub.status.idle":"2022-02-05T21:42:56.523155Z","shell.execute_reply.started":"2022-02-05T21:42:56.51353Z","shell.execute_reply":"2022-02-05T21:42:56.522489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detoxify _multilingual\ndef f3 (df,target_col = \"text\"):\n    huggingface_config_path = '../input/bert-base-uncased'\n    d =  Detoxify('multilingual', device='cuda')\n\n    uniqeu_text_list = df[target_col].unique()\n    uniqeu_text_list = list(uniqeu_text_list)\n    \n    toxic_list = []\n    for text in tqdm(uniqeu_text_list):\n        result = d.predict(text[:512])\n        toxic_list.append(result)\n        \n    temp = pd.DataFrame()\n    temp[target_col] = uniqeu_text_list\n    temp[\"di\"] = toxic_list\n    cols = toxic_list[0].keys()\n    \n    for col in cols:\n        temp[col + \"_multilingual\"] = temp[\"di\"].apply(lambda x:x[col])\n    temp = temp.drop([\"di\"],axis = 1)\n    df = pd.merge(df,temp,on = target_col,suffixes = [\"\",\"_multilingual\"])\n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:42:56.526088Z","iopub.execute_input":"2022-02-05T21:42:56.526278Z","iopub.status.idle":"2022-02-05T21:42:56.53483Z","shell.execute_reply.started":"2022-02-05T21:42:56.526256Z","shell.execute_reply":"2022-02-05T21:42:56.534089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf-idf\nimport umap.umap_ as umap\n\ndef f4 (df,target_col = \"text\"):\n    tfvec = TfidfVectorizer(min_df= 3, max_df=0.5, analyzer = 'char_wb', ngram_range = (3,5))\n    tfv = tfvec.fit_transform(df[target_col])\n    print(\"vec is end\")\n    model_tsne = umap.UMAP(n_components=10, n_neighbors=10)\n    result = model_tsne.fit_transform(tfv)\n    l = len(result[0])\n    for i in range(l):\n        df[f\"tf_idf_{i}\"] = result[:,i]\n    \n    return df\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T22:02:16.150668Z","iopub.execute_input":"2022-02-05T22:02:16.151398Z","iopub.status.idle":"2022-02-05T22:02:16.156554Z","shell.execute_reply.started":"2022-02-05T22:02:16.15136Z","shell.execute_reply":"2022-02-05T22:02:16.155851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean(data, col):\n\n    # Clean some punctutations\n    data[col] = data[col].str.replace('\\n', ' \\n ')\n    # Remove ip address\n    data[col] = data[col].str.replace(r'(([0-9]+\\.){2,}[0-9]+)',' ')\n    \n    data[col] = data[col].str.replace(r'([a-zA-Z]+)([/!?.])([a-zA-Z]+)',r'\\1 \\2 \\3')\n    # Replace repeating characters more than 3 times to length of 3\n    data[col] = data[col].str.replace(r'([*!?\\'])\\1\\1{2,}',r'\\1\\1\\1')\n    # patterns with repeating characters \n    data[col] = data[col].str.replace(r'([a-zA-Z])\\1{2,}\\b',r'\\1\\1')\n    data[col] = data[col].str.replace(r'([a-zA-Z])\\1\\1{2,}\\B',r'\\1\\1\\1')\n    data[col] = data[col].str.replace(r'[ ]{2,}',' ').str.strip()   \n    # Add space around repeating characters\n    data[col] = data[col].str.replace(r'([*!?\\']+)',r' \\1 ')    \n    \n    return data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# some basic features\ndef f5(df,target_col = \"text\"):\n    df[\"length\"] = df[target_col].apply(len)\n    df[\"word_size\"] = df[target_col].apply(lambda x:len(x.split()))\n    \n    for c in [\"!\",\"?\",\"！\",\"？\"]:\n        df[f\"count_{c}\"] = df[\"text\"].apply(lambda x:x.count(c)) \n    \n    df[\"cleaned_text\"] = df[\"text\"]\n    df = clean(df,\"cleaned_text\")\n    df[\"cleaned_length\"] = df[\"cleaned_text\"].apply(len)\n    df[\"cleaned_word_size\"] = df[\"cleaned_text\"].apply(lambda x:len(x.split()))\n    for col in [\"length\",\"word_size\",\"cleaned_length\",\"cleaned_word_size\"]:\n        for col2 in [\"length\",\"word_size\",\"cleaned_length\",\"cleaned_word_size\"]:\n            if col > col2:\n                df[f\"{col}_{col2}_div\"]  = df[col]/(df[col2] + 1)\n    df = df.drop(\"cleaned_text\",axis = 1)\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = f1(df)\ndf = f2(df)\ndf = f3(df)\n\ntest = f1(test)\ntest = f2(test)\ntest = f3(test)\n\n# df = f5(df)\n# test = f5(test)\n\n\n\ntemp = pd.DataFrame()\ntemp[\"text\"] = list(df[\"text\"]) + list(test[\"text\"])\n\n\ntemp = f4(temp)\ntemp_train = temp[:len(df)]\ntemp_test = temp[len(df):]\n\nfor col in temp_train.columns:\n    if \"tf_idf_\" in col:\n        df[col] = list(temp_train[col])\nfor col in temp_test.columns:\n    if \"tf_idf_\" in col:\n        test[col] = list(temp_test[col])\n\n\ndf_sub = test","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:12:13.316628Z","iopub.execute_input":"2022-02-05T20:12:13.317443Z","iopub.status.idle":"2022-02-05T20:12:13.505404Z","shell.execute_reply.started":"2022-02-05T20:12:13.317392Z","shell.execute_reply":"2022-02-05T20:12:13.503712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:22.223883Z","iopub.execute_input":"2022-02-05T17:39:22.224525Z","iopub.status.idle":"2022-02-05T17:39:22.231083Z","shell.execute_reply.started":"2022-02-05T17:39:22.224485Z","shell.execute_reply":"2022-02-05T17:39:22.23023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:39:25.70801Z","iopub.execute_input":"2022-02-05T17:39:25.708314Z","iopub.status.idle":"2022-02-05T17:39:25.716166Z","shell.execute_reply.started":"2022-02-05T17:39:25.708274Z","shell.execute_reply":"2022-02-05T17:39:25.715328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ridge feature","metadata":{}},{"cell_type":"code","source":"n_folds = 2\n\nfrac_1 = 0.7\nfrac_1_factor = 1.3","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:12:45.036101Z","iopub.execute_input":"2022-02-05T20:12:45.036393Z","iopub.status.idle":"2022-02-05T20:12:45.040993Z","shell.execute_reply.started":"2022-02-05T20:12:45.036361Z","shell.execute_reply":"2022-02-05T20:12:45.039943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def timer(func):\n    def wrapper(*args, **kws):\n        st = time.time()\n        res = func(*args, **kws)\n        et = time.time()\n        tt = (et-st)/60\n        print(f'Time taken is {tt:.2f} mins')\n        return res\n    return wrapper","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:12:45.222403Z","iopub.execute_input":"2022-02-05T20:12:45.222702Z","iopub.status.idle":"2022-02-05T20:12:45.229259Z","shell.execute_reply.started":"2022-02-05T20:12:45.222669Z","shell.execute_reply":"2022-02-05T20:12:45.228339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass LengthTransformer(BaseEstimator, TransformerMixin):\n\n    def fit(self, X, y=None):\n        return self\n    def transform(self, X):\n        return sparse.csr_matrix([[(len(x)-360)/550] for x in X])\n    def get_feature_names(self):\n        return [\"lngth\"]\n\nclass LengthUpperTransformer(BaseEstimator, TransformerMixin):\n\n    def fit(self, X, y=None):\n        return self\n    def transform(self, X):\n        return sparse.csr_matrix([[int(sum([1 for y in x if y.isupper()])/len(x) > 0.75) ] for x in X])\n    def get_feature_names(self):\n        return [\"lngth_uppercase\"]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:12:45.642009Z","iopub.execute_input":"2022-02-05T20:12:45.642716Z","iopub.status.idle":"2022-02-05T20:12:45.651317Z","shell.execute_reply.started":"2022-02-05T20:12:45.642627Z","shell.execute_reply":"2022-02-05T20:12:45.650484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ridge regression to make feature","metadata":{}},{"cell_type":"code","source":"\ndef train_pipeline(pipeline, data_path_name, n_folds, train_df, test_df,clean_prm = False):\n    \n    print(\"train_df size:\",len(train_df))\n    print(\"test_df size:\",len(test_df))\n    train_result_df = pd.DataFrame()\n    test_result_df = pd.DataFrame()\n    \n    train_result_df[\"text\"] = train_df[\"text\"]\n    test_result_df[\"text\"] = test_df[\"text\"]\n    for fld in range(n_folds):\n        print(\"\\n\\n\")\n        print(f' ****************************** FOLD: {fld} ******************************')\n        df = pd.read_csv(f'../input/mega-b-ridge-to-the-top-lb-0-842/{data_path_name}_fld{fld}.csv')\n        # Train the pipeline\n        pipeline.fit(df['text'], df['y'])\n\n        # What are the important features for toxicity\n\n    #     print('\\nTotal number of features:', len(pipeline['features'].get_feature_names()) )\n\n    #     feature_wts = sorted(list(zip(pipeline['features'].get_feature_names(), \n    #                                   np.round(pipeline['clf'].coef_,2) )), \n    #                          key = lambda x:x[1], \n    #                          reverse=True)\n\n    #     display(pd.DataFrame(feature_wts[:50], columns = ['feat','val']).T)\n        #.plot('feat','val',kind='barh',figsize = (8,8) )\n        #plt.show()\n\n        if clean_prm:\n            print(\"\\npredict validation data \")\n            train_result_df[f\"{fld}_{data_path_name}\"] = pipeline.predict(clean(train_df,'text')['text'])\n            print(\"\\npredict test data \")\n            test_result_df[f\"{fld}_{data_path_name}\"] = pipeline.predict(clean(test_df,'text')['text'])\n        else:\n            print(\"\\npredict validation data \")\n            train_result_df[f\"{fld}_{data_path_name}\"] = pipeline.predict(train_df['text'])\n            test_result_df[f\"{fld}_{data_path_name}\"] = pipeline.predict(test_df['text'])\n    \n    #         val_preds_arr1_tmp[:,fld] = pipeline.predict(df_val['text'])\n    #         test_preds_arr_tmp[:,fld] = pipeline.predict(df_sub['text'])\n    \n    print(\"train_result_df size:\",len(train_result_df))\n    print(\"test_result_df size:\",len(test_result_df))\n    return train_result_df,test_result_df","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:30:41.304203Z","iopub.execute_input":"2022-02-05T20:30:41.304914Z","iopub.status.idle":"2022-02-05T20:30:41.315642Z","shell.execute_reply.started":"2022-02-05T20:30:41.304871Z","shell.execute_reply":"2022-02-05T20:30:41.314859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nname_list = [\"tweet\",\"wikipedia\",\"dfm\",\"df_clean\",\"df2\",\"malignant\"]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:30:42.67354Z","iopub.execute_input":"2022-02-05T20:30:42.674041Z","iopub.status.idle":"2022-02-05T20:30:42.677679Z","shell.execute_reply.started":"2022-02-05T20:30:42.674006Z","shell.execute_reply":"2022-02-05T20:30:42.676976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred_list1 = []\nval_pred_list2 = []\ntest_pred_list = []","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:30:43.473025Z","iopub.execute_input":"2022-02-05T20:30:43.473639Z","iopub.status.idle":"2022-02-05T20:30:43.477472Z","shell.execute_reply.started":"2022-02-05T20:30:43.473596Z","shell.execute_reply":"2022-02-05T20:30:43.476706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ntrain_feature_list = []\ntest_feature_list = []\nfor name in tqdm(name_list):\n    print(\"use data name is : \", name)\n#     if name == \"df2\":\n#         break\n    features = FeatureUnion([\n        #('vect1', LengthTransformer()),\n        #('vect2', LengthUpperTransformer()),\n        (\"vect3\", TfidfVectorizer(min_df= 3, max_df=0.5, \n                                  analyzer = 'char_wb', ngram_range = (3,5))),\n        #(\"vect4\", TfidfVectorizer(min_df= 5, max_df=0.5, analyzer = 'word', token_pattern=r'(?u)\\b\\w{8,}\\b')),\n\n    ])\n    pipeline = Pipeline(\n        [\n            (\"features\", features),\n            #(\"clf\", RandomForestRegressor(n_estimators = 5, min_sample_leaf=3)),\n            (\"clf\", Ridge()),\n            #(\"clf\",LinearRegression())\n        ]\n    )\n\n    is_clean = True if \"clean\" in name else False\n    df_copy = df.copy()\n    df_sub_copy = df_sub.copy()\n    train_f,test_f = train_pipeline(pipeline, name, n_folds, df_copy, df_sub_copy, clean_prm=is_clean)\n    train_feature_list.append(train_f)\n    test_feature_list.append(test_f)\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:30:51.130649Z","iopub.execute_input":"2022-02-05T20:30:51.130923Z","iopub.status.idle":"2022-02-05T20:34:20.997538Z","shell.execute_reply.started":"2022-02-05T20:30:51.130897Z","shell.execute_reply":"2022-02-05T20:34:20.996736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for temp in train_feature_list:\n    df = pd.merge(df,temp,on = \"text\")\n\n\nfor temp in test_feature_list:\n    df_sub = pd.merge(df_sub,temp,on = \"text\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:40:43.063248Z","iopub.execute_input":"2022-02-05T20:40:43.063834Z","iopub.status.idle":"2022-02-05T20:40:43.094945Z","shell.execute_reply.started":"2022-02-05T20:40:43.063799Z","shell.execute_reply":"2022-02-05T20:40:43.094476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nkf = KFold(n_splits=5,shuffle = True,random_state = 773105)\ny = df[\"total_score\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:45:37.809418Z","iopub.execute_input":"2022-02-05T20:45:37.81035Z","iopub.status.idle":"2022-02-05T20:45:37.81559Z","shell.execute_reply.started":"2022-02-05T20:45:37.810294Z","shell.execute_reply":"2022-02-05T20:45:37.814821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import lightgbm as lgb\nfrom sklearn.metrics import mean_squared_log_error\nfrom tqdm import tqdm\n\nimport sklearn  as sk\n\nparams = {\n 'boosting_type': 'gbdt',\n    'objective': 'rmse',\n    'seed': 773105,\n    'learning_rate': 0.05,\n    \"n_jobs\": -1,\n    \"verbose\": -1,\n    'n_estimators': 10000,     \n    'max_depth': 7,\n     'importance_type': 'gain',  \n\n\n}","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:45:38.95759Z","iopub.execute_input":"2022-02-05T20:45:38.957853Z","iopub.status.idle":"2022-02-05T20:45:39.778355Z","shell.execute_reply.started":"2022-02-05T20:45:38.957826Z","shell.execute_reply":"2022-02-05T20:45:39.777662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_cols = [\"text\",\"total_score\"]","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:45:41.314291Z","iopub.execute_input":"2022-02-05T20:45:41.314948Z","iopub.status.idle":"2022-02-05T20:45:41.318824Z","shell.execute_reply.started":"2022-02-05T20:45:41.314902Z","shell.execute_reply":"2022-02-05T20:45:41.318092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictor_list = []\n\nfor i,(train_index, valid_index) in enumerate(kf.split(y)):\n    print(\"train_index:\", train_index, \"test_index:\", valid_index)\n    train_x = df.iloc[train_index].drop(remove_cols,axis = 1)\n    valid_x = df.iloc[valid_index].drop(remove_cols,axis = 1)\n    \n    train_y = y.iloc[train_index]\n    valid_y = y.iloc[valid_index]\n\n    lgb_train = lgb.Dataset(train_x, train_y )\n    lgb_eval = lgb.Dataset(valid_x,  valid_y , reference=lgb_train)\n\n\n    predeictor = lgb.train(params,\n                    lgb_train,\n                    num_boost_round=500,\n                    valid_sets=[lgb_eval, lgb_train],\n                    verbose_eval=10,\n                    early_stopping_rounds = 1000\n\n                    )\n    predeictor.save_model(f\"model_{i}\")\n    predictor_list.append(predeictor)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:45:42.175717Z","iopub.execute_input":"2022-02-05T20:45:42.176482Z","iopub.status.idle":"2022-02-05T20:45:43.300207Z","shell.execute_reply.started":"2022-02-05T20:45:42.176414Z","shell.execute_reply":"2022-02-05T20:45:43.299408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.stats import rankdata\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:22:04.550997Z","iopub.execute_input":"2022-02-05T21:22:04.551266Z","iopub.status.idle":"2022-02-05T21:22:04.554662Z","shell.execute_reply.started":"2022-02-05T21:22:04.551234Z","shell.execute_reply":"2022-02-05T21:22:04.554005Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame()\nfor i,p in enumerate(predictor_list):\n    sub[i] = p.predict(df_sub[train_x.columns])\nsub[\"score\"] = sub.sum(axis = 1)\nsub[\"score\"] = rankdata(sub[\"score\"], method='ordinal')","metadata":{"execution":{"iopub.status.busy":"2022-02-05T21:21:42.357113Z","iopub.execute_input":"2022-02-05T21:21:42.357632Z","iopub.status.idle":"2022-02-05T21:21:42.437892Z","shell.execute_reply.started":"2022-02-05T21:21:42.357541Z","shell.execute_reply":"2022-02-05T21:21:42.436748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[\"comment_id\"] = df_sub[\"comment_id\"]\n","metadata":{"execution":{"iopub.status.busy":"2022-02-05T20:46:27.375993Z","iopub.execute_input":"2022-02-05T20:46:27.376699Z","iopub.status.idle":"2022-02-05T20:46:27.381195Z","shell.execute_reply.started":"2022-02-05T20:46:27.376666Z","shell.execute_reply":"2022-02-05T20:46:27.380472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub[[\"comment_id\",\"score\"]].to_csv(\"submission.csv\",index = None)","metadata":{"execution":{"iopub.status.busy":"2022-01-29T08:11:09.876552Z","iopub.status.idle":"2022-01-29T08:11:09.87718Z","shell.execute_reply.started":"2022-01-29T08:11:09.876934Z","shell.execute_reply":"2022-01-29T08:11:09.876957Z"},"trusted":true},"execution_count":null,"outputs":[]}]}