{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n!pip install pymorphy2\n!pip install pyspellchecker\nimport os, pandas as pd, re, numpy as np, pymorphy2, ast, nltk, math\nfrom nltk.tokenize import word_tokenize\nfrom nltk.tokenize.toktok import ToktokTokenizer\nfrom nltk.stem import PorterStemmer\nfrom spellchecker import SpellChecker\nfrom catboost import Pool, CatBoostRegressor\nfrom sklearn.metrics import mean_absolute_error\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\nspell = SpellChecker()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-14T22:01:43.494778Z","iopub.execute_input":"2021-12-14T22:01:43.495564Z","iopub.status.idle":"2021-12-14T22:02:05.55992Z","shell.execute_reply.started":"2021-12-14T22:01:43.495443Z","shell.execute_reply":"2021-12-14T22:02:05.559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\ntrain1 = train1[['id', 'comment_text', 'toxic']]\ntrain1.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:51:04.460994Z","iopub.execute_input":"2021-12-13T21:51:04.461824Z","iopub.status.idle":"2021-12-13T21:51:07.101107Z","shell.execute_reply.started":"2021-12-13T21:51:04.461785Z","shell.execute_reply":"2021-12-13T21:51:07.100382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1['toxic'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:51:07.102566Z","iopub.execute_input":"2021-12-13T21:51:07.10278Z","iopub.status.idle":"2021-12-13T21:51:07.108987Z","shell.execute_reply.started":"2021-12-13T21:51:07.102755Z","shell.execute_reply":"2021-12-13T21:51:07.108194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv')\ntrain2 = train2[['id', 'comment_text', 'toxic']]\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:51:07.110238Z","iopub.execute_input":"2021-12-13T21:51:07.110757Z","iopub.status.idle":"2021-12-13T21:51:23.790614Z","shell.execute_reply.started":"2021-12-13T21:51:07.11072Z","shell.execute_reply":"2021-12-13T21:51:23.789692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2['toxic'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:51:23.793964Z","iopub.execute_input":"2021-12-13T21:51:23.794766Z","iopub.status.idle":"2021-12-13T21:51:23.827271Z","shell.execute_reply.started":"2021-12-13T21:51:23.794722Z","shell.execute_reply":"2021-12-13T21:51:23.826306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train2[train2.toxic != 0])","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:51:44.33342Z","iopub.execute_input":"2021-12-13T21:51:44.333675Z","iopub.status.idle":"2021-12-13T21:51:44.404394Z","shell.execute_reply.started":"2021-12-13T21:51:44.333648Z","shell.execute_reply":"2021-12-13T21:51:44.403625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Preprocessing:\n    def __init__(self, stopwords):\n        self.rgc = re.compile('[^a-zа-яё0-9-_]')\n        self.tokenizer = ToktokTokenizer()\n        self.lemmatizer = pymorphy2.MorphAnalyzer()\n        self.stemmer = PorterStemmer()\n\n        with open(stopwords, 'r') as f:\n            self.stopwords = set(f.read().split('\\n'))\n\n    def preproc(self, text, check_stopwords=True, check_length=True, use_lemm=False, use_stem=True):\n        s = re.sub(\"\\n\", r\" \", text)\n        \n        s = re.sub(r'\\w*\\d\\w*', '', s).strip() #remove digits\n        s = re.sub(' +', ' ', s) # join\n        s = re.sub(r'(.)\\1+', r'\\1\\1', s) # remove repetitions\n        \n        s = re.sub(\"'\", r\" \", s)\n        s = s.lower()\n        s = self.rgc.sub(\" \", s)\n\n        final_agg = []\n        tf = {}\n\n        for i, token in enumerate(self.tokenizer.tokenize(s)):\n            if check_length and len(token) < 2:\n                continue\n            if token[-1] == '-' or token[0] == '-':\n                continue\n            if use_lemm:\n                token = self.lemmatizer.parse(token)[0].normal_form\n            if use_stem:\n                token = self.stemmer.stem(token)\n            if token not in self.stopwords or not check_stopwords:\n                if token not in tf:\n                    tf[token] = 0\n                tf[token] += 1\n                final_agg.append(token)\n\n        return ' '.join(final_agg), tf\n    \ndef prepare_df(df):\n    df['len_comment'] = df.comment_text.apply(lambda t: len(t))\n    df['comment_text_proc'] = df['comment_text'].apply(p.preproc)\n    df = df.drop(columns=['comment_text'])\n    \n    df['com_tf'] = pd.DataFrame(df['comment_text_proc'].tolist(), index=df.index)[1] \n    def right_tf(dic):\n        if sum(list(dic.values())) != 0:\n            right_tf = list(np.array(list(dic.values())) * ( 1 / sum(list(dic.values()))))\n        else:\n            right_tf = list(np.array(list(dic.values())) * 0)\n        keys = list(dic.keys())\n        dic = dict(zip(keys, right_tf))\n        return dic\n    df['com_tf'] = df['com_tf'].apply(right_tf)\n    \n    df.drop(columns=['comment_text_proc'])\n    df = df.reset_index()\n    \n    return df\n    \ndef read_proc_save(path_read, path_save_train, path_save_test):\n    #train = pd.read_csv('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\n    train = pd.read_csv(path_read)\n    train = train[['id', 'comment_text', 'toxic']]\n    print(train['toxic'].unique())\n    \n    if path_read == '../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv':\n        train = train[train.toxic != 0]\n    \n    per_train = 0.7\n    df_tr = train.sample(frac=per_train)\n    df_ts = train.merge(df_tr, how = 'outer' ,indicator=True).loc[lambda x : x['_merge']=='left_only']\n    df_ts = df_ts.drop(columns=['_merge'])\n    \n    df_tr = prepare_df(df_tr)\n    print('prepared train')\n    df_ts = prepare_df(df_ts)\n    print('prepared test')\n    df_tr.to_csv(path_save_train)\n    print('saved train')\n    df_ts.to_csv(path_save_test)\n    print('saved test')\n    \ndef get_freq_dict(freq_string):\n    return ast.literal_eval(freq_string[freq_string.find(',')+2:-1])\n\ndef get_comment(freq_string):\n    return freq_string[2:freq_string.find('{') - 3]\n\ndef final_proc(train):\n    train['freq_dict'] = train.comment_text_proc.apply(get_freq_dict)\n    train['comment'] = train.comment_text_proc.apply(get_comment)\n    train.com_tf = train.com_tf.apply(lambda t: ast.literal_eval(t))\n    train = train.drop(columns=['comment_text_proc'])\n    return train\n  \np = Preprocessing(stopwords='../input/stoppp/stopwords.txt')","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:05.561995Z","iopub.execute_input":"2021-12-14T22:02:05.562292Z","iopub.status.idle":"2021-12-14T22:02:05.91052Z","shell.execute_reply.started":"2021-12-14T22:02:05.562252Z","shell.execute_reply":"2021-12-14T22:02:05.90963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"read_proc_save('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv', \n               'train_data_1.csv', 'test_data_1.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:23:14.817057Z","iopub.execute_input":"2021-12-13T21:23:14.817261Z","iopub.status.idle":"2021-12-13T21:31:36.443669Z","shell.execute_reply.started":"2021-12-13T21:23:14.81723Z","shell.execute_reply":"2021-12-13T21:31:36.44273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"read_proc_save('../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv', \n               'train_data_2.csv', 'test_data_2.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-13T21:52:48.806215Z","iopub.execute_input":"2021-12-13T21:52:48.806616Z","iopub.status.idle":"2021-12-13T22:10:56.469474Z","shell.execute_reply.started":"2021-12-13T21:52:48.806571Z","shell.execute_reply":"2021-12-13T22:10:56.468557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv('../input/processed/train_data_1.csv').drop(columns=['Unnamed: 0', 'id'])\ntrain1.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:05.911867Z","iopub.execute_input":"2021-12-14T22:02:05.912724Z","iopub.status.idle":"2021-12-14T22:02:09.634596Z","shell.execute_reply.started":"2021-12-14T22:02:05.912656Z","shell.execute_reply":"2021-12-14T22:02:09.63375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1 = pd.read_csv('../input/processed/test_data_1.csv').drop(columns=['Unnamed: 0', 'id'])\ntest1.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:09.636481Z","iopub.execute_input":"2021-12-14T22:02:09.636708Z","iopub.status.idle":"2021-12-14T22:02:11.187316Z","shell.execute_reply.started":"2021-12-14T22:02:09.636682Z","shell.execute_reply":"2021-12-14T22:02:11.186289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = pd.read_csv('../input/processed/train_data_2.csv').drop(columns=['Unnamed: 0', 'id'])\ntrain2.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:11.188746Z","iopub.execute_input":"2021-12-14T22:02:11.188952Z","iopub.status.idle":"2021-12-14T22:02:18.89814Z","shell.execute_reply.started":"2021-12-14T22:02:11.188927Z","shell.execute_reply":"2021-12-14T22:02:18.89719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test2 = pd.read_csv('../input/processed/test_data_2.csv').drop(columns=['Unnamed: 0', 'id'])\ntest2.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:18.899335Z","iopub.execute_input":"2021-12-14T22:02:18.899568Z","iopub.status.idle":"2021-12-14T22:02:22.199711Z","shell.execute_reply.started":"2021-12-14T22:02:18.899538Z","shell.execute_reply":"2021-12-14T22:02:22.198735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = final_proc(train1)\ntest1 = final_proc(test1)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:22.201036Z","iopub.execute_input":"2021-12-14T22:02:22.201242Z","iopub.status.idle":"2021-12-14T22:02:56.40153Z","shell.execute_reply.started":"2021-12-14T22:02:22.201217Z","shell.execute_reply":"2021-12-14T22:02:56.40056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:56.404752Z","iopub.execute_input":"2021-12-14T22:02:56.405161Z","iopub.status.idle":"2021-12-14T22:02:56.430375Z","shell.execute_reply.started":"2021-12-14T22:02:56.405124Z","shell.execute_reply":"2021-12-14T22:02:56.429425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:56.431842Z","iopub.execute_input":"2021-12-14T22:02:56.432145Z","iopub.status.idle":"2021-12-14T22:02:56.462754Z","shell.execute_reply.started":"2021-12-14T22:02:56.432042Z","shell.execute_reply":"2021-12-14T22:02:56.461731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = final_proc(train2)\ntest2 = final_proc(test2)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:02:56.464992Z","iopub.execute_input":"2021-12-14T22:02:56.465791Z","iopub.status.idle":"2021-12-14T22:04:21.894425Z","shell.execute_reply.started":"2021-12-14T22:02:56.465753Z","shell.execute_reply":"2021-12-14T22:04:21.89375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:04:21.895822Z","iopub.execute_input":"2021-12-14T22:04:21.896853Z","iopub.status.idle":"2021-12-14T22:04:21.921517Z","shell.execute_reply.started":"2021-12-14T22:04:21.896803Z","shell.execute_reply":"2021-12-14T22:04:21.920633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test2","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:04:21.923147Z","iopub.execute_input":"2021-12-14T22:04:21.923762Z","iopub.status.idle":"2021-12-14T22:04:21.947157Z","shell.execute_reply.started":"2021-12-14T22:04:21.923683Z","shell.execute_reply":"2021-12-14T22:04:21.94595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_most_common_words(df_tr, num_most_com=1000):\n    all_words = ''\n    for i in range(len(df_tr)):\n        all_words += df_tr.comment[i] + ' '\n    all_words = all_words.split()\n    all_words_dist = nltk.FreqDist(w for w in all_words)\n    most_common = all_words_dist.most_common(num_most_com)\n    most_common_words = [item[0] for item in most_common]\n    return most_common_words","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:05:12.592439Z","iopub.execute_input":"2021-12-14T22:05:12.59301Z","iopub.status.idle":"2021-12-14T22:05:12.598103Z","shell.execute_reply.started":"2021-12-14T22:05:12.592972Z","shell.execute_reply":"2021-12-14T22:05:12.597097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mc_1 = get_most_common_words(train1)\nmc_2 = get_most_common_words(train2)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:05:14.885641Z","iopub.execute_input":"2021-12-14T22:05:14.885948Z","iopub.status.idle":"2021-12-14T22:05:33.567702Z","shell.execute_reply.started":"2021-12-14T22:05:14.885915Z","shell.execute_reply":"2021-12-14T22:05:33.566129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:23:32.370544Z","iopub.execute_input":"2021-12-14T22:23:32.370894Z","iopub.status.idle":"2021-12-14T22:23:32.3977Z","shell.execute_reply.started":"2021-12-14T22:23:32.370859Z","shell.execute_reply":"2021-12-14T22:23:32.397057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mc_2","metadata":{"execution":{"iopub.status.busy":"2021-12-14T22:05:49.23718Z","iopub.execute_input":"2021-12-14T22:05:49.237522Z","iopub.status.idle":"2021-12-14T22:05:49.256413Z","shell.execute_reply.started":"2021-12-14T22:05:49.237475Z","shell.execute_reply":"2021-12-14T22:05:49.255511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('../input/most-common-words/mc_1.txt', 'r') as f:\n    mc_1 = set(f.read().split())\nwith open('../input/most-common-words/mc_2.txt', 'r') as f:\n    mc_2 = set(f.read().split())","metadata":{"execution":{"iopub.status.busy":"2021-12-14T18:44:44.673586Z","iopub.execute_input":"2021-12-14T18:44:44.674296Z","iopub.status.idle":"2021-12-14T18:44:44.693661Z","shell.execute_reply.started":"2021-12-14T18:44:44.674261Z","shell.execute_reply":"2021-12-14T18:44:44.692478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reg_df_prep(df, most_common_words):\n    all_tf_dicts = [df['freq_dict'][i] for i in range(len(df))]\n    \n    def get_word_freq(word):\n        freq = 0\n        for i in range(len(df)):\n            if word in all_tf_dicts[i].keys():\n                freq += 1\n        return freq\n    \n    i = 0 #\n    for word in most_common_words:\n        df[word] = np.array([d.get(word) for d in df['com_tf']])\n        df[word] = df[word].fillna(0)\n        if get_word_freq(word) == 0:\n            df[word] = [0] * len(df)\n            print(word)\n        else:   \n            df[word] = math.log10(df.shape[0]/get_word_freq(word)) * df[word]\n        \n        i += 1\n        if i%100 == 0:\n            print(i) #\n        \n    df = df.drop(columns=['com_tf', 'freq_dict', 'index'])\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-12-14T18:44:44.694898Z","iopub.execute_input":"2021-12-14T18:44:44.695184Z","iopub.status.idle":"2021-12-14T18:44:44.705668Z","shell.execute_reply.started":"2021-12-14T18:44:44.695136Z","shell.execute_reply":"2021-12-14T18:44:44.704972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr_1 = reg_df_prep(train1, mc_1)\ndf_ts_1 = reg_df_prep(test1, mc_1)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T17:30:59.259544Z","iopub.execute_input":"2021-12-14T17:30:59.260156Z","iopub.status.idle":"2021-12-14T17:38:39.995057Z","shell.execute_reply.started":"2021-12-14T17:30:59.260083Z","shell.execute_reply":"2021-12-14T17:38:39.993978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ts_1.shape, df_tr_1.shape, len(mc_1)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T17:54:53.377724Z","iopub.execute_input":"2021-12-14T17:54:53.378081Z","iopub.status.idle":"2021-12-14T17:54:53.384302Z","shell.execute_reply.started":"2021-12-14T17:54:53.378039Z","shell.execute_reply":"2021-12-14T17:54:53.383676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr_2 = reg_df_prep(train2, mc_2)\ndf_ts_2 = reg_df_prep(test2, mc_2)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T18:44:44.707726Z","iopub.execute_input":"2021-12-14T18:44:44.708613Z","iopub.status.idle":"2021-12-14T18:59:41.722479Z","shell.execute_reply.started":"2021-12-14T18:44:44.708563Z","shell.execute_reply":"2021-12-14T18:59:41.720826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ts_2.shape, df_tr_2.shape, len(mc_2) \n","metadata":{"execution":{"iopub.status.busy":"2021-12-14T18:59:41.724723Z","iopub.execute_input":"2021-12-14T18:59:41.725045Z","iopub.status.idle":"2021-12-14T18:59:41.735591Z","shell.execute_reply.started":"2021-12-14T18:59:41.725Z","shell.execute_reply":"2021-12-14T18:59:41.734345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_tr_2.to_csv('final_train_2.csv')\ndf_ts_2.to_csv('final_test_2.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-14T18:59:41.736977Z","iopub.execute_input":"2021-12-14T18:59:41.737242Z","iopub.status.idle":"2021-12-14T19:08:02.984254Z","shell.execute_reply.started":"2021-12-14T18:59:41.737213Z","shell.execute_reply":"2021-12-14T19:08:02.983143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv('../input/final-data-1/final_train_1.csv').drop(columns=['Unnamed: 0'])","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:30:38.406726Z","iopub.execute_input":"2021-12-14T19:30:38.406952Z","iopub.status.idle":"2021-12-14T19:31:08.143551Z","shell.execute_reply.started":"2021-12-14T19:30:38.406925Z","shell.execute_reply":"2021-12-14T19:31:08.14273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1 = pd.read_csv('../input/final-data-1/final_test_1.csv').drop(columns=['Unnamed: 0'])","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:31:08.144901Z","iopub.execute_input":"2021-12-14T19:31:08.145106Z","iopub.status.idle":"2021-12-14T19:31:20.144791Z","shell.execute_reply.started":"2021-12-14T19:31:08.145082Z","shell.execute_reply":"2021-12-14T19:31:20.14407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train2 = pd.read_csv('../input/finaldata1/final_train_2.csv').drop(columns=['Unnamed: 0'])","metadata":{"execution":{"iopub.status.busy":"2021-12-14T17:16:10.872231Z","iopub.execute_input":"2021-12-14T17:16:10.872661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test2 = pd.read_csv('../input/finaldata1/final_test_2.csv').drop(columns=['Unnamed: 0'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1.shape","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:31:31.651444Z","iopub.execute_input":"2021-12-14T19:31:31.651812Z","iopub.status.idle":"2021-12-14T19:31:31.660479Z","shell.execute_reply.started":"2021-12-14T19:31:31.651743Z","shell.execute_reply":"2021-12-14T19:31:31.659551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1.shape","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:31:33.921161Z","iopub.execute_input":"2021-12-14T19:31:33.921427Z","iopub.status.idle":"2021-12-14T19:31:33.92765Z","shell.execute_reply.started":"2021-12-14T19:31:33.921399Z","shell.execute_reply":"2021-12-14T19:31:33.927044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test1.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T09:03:55.437251Z","iopub.execute_input":"2021-12-14T09:03:55.4378Z","iopub.status.idle":"2021-12-14T09:03:55.490113Z","shell.execute_reply.started":"2021-12-14T09:03:55.437757Z","shell.execute_reply":"2021-12-14T09:03:55.489235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr_1 = train1.drop(columns=['toxic', 'comment']).to_numpy()\nX_ts_1 = test1.drop(columns=['toxic', 'comment']).to_numpy()\ny_tr_1 = train1[['toxic']].to_numpy()\ny_ts_1 = test1[['toxic']].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:31:41.553012Z","iopub.execute_input":"2021-12-14T19:31:41.553524Z","iopub.status.idle":"2021-12-14T19:31:44.654126Z","shell.execute_reply.started":"2021-12-14T19:31:41.553457Z","shell.execute_reply":"2021-12-14T19:31:44.652225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr_1","metadata":{"execution":{"iopub.status.busy":"2021-12-14T09:04:05.82332Z","iopub.execute_input":"2021-12-14T09:04:05.823599Z","iopub.status.idle":"2021-12-14T09:04:05.831349Z","shell.execute_reply.started":"2021-12-14T09:04:05.823571Z","shell.execute_reply":"2021-12-14T09:04:05.830732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pool = Pool(X_tr_1, y_tr_1)\ntest_pool = Pool(X_ts_1)\n\nmodel = CatBoostRegressor(iterations=100, depth=8, learning_rate=0.1, loss_function='MAE')\nmodel.fit(train_pool)\n\n'''preds = model.predict(test_pool)\npreds = np.clip(preds, 0, 10)\n\nmean_absolute_error(y_ts, preds)'''","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:36:47.638777Z","iopub.execute_input":"2021-12-14T19:36:47.639296Z","iopub.status.idle":"2021-12-14T19:37:51.087619Z","shell.execute_reply.started":"2021-12-14T19:36:47.639261Z","shell.execute_reply":"2021-12-14T19:37:51.086616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_pool)\npreds = np.clip(preds, 0, 1)\npreds[preds >= 0.5] = 1\npreds[preds < 0.5] = 0\npreds","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:49:40.181339Z","iopub.execute_input":"2021-12-14T19:49:40.181716Z","iopub.status.idle":"2021-12-14T19:49:40.23126Z","shell.execute_reply.started":"2021-12-14T19:49:40.181658Z","shell.execute_reply":"2021-12-14T19:49:40.230518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.sum()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:51:38.462497Z","iopub.execute_input":"2021-12-14T19:51:38.462879Z","iopub.status.idle":"2021-12-14T19:51:38.470651Z","shell.execute_reply.started":"2021-12-14T19:51:38.462844Z","shell.execute_reply":"2021-12-14T19:51:38.469737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_ts_1 = np.squeeze(y_ts_1)\ny_ts_1.sum()","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:51:33.015152Z","iopub.execute_input":"2021-12-14T19:51:33.015603Z","iopub.status.idle":"2021-12-14T19:51:33.022068Z","shell.execute_reply.started":"2021-12-14T19:51:33.015552Z","shell.execute_reply":"2021-12-14T19:51:33.021219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_absolute_error(y_ts_1, preds)","metadata":{"execution":{"iopub.status.busy":"2021-12-14T19:53:42.110023Z","iopub.execute_input":"2021-12-14T19:53:42.110691Z","iopub.status.idle":"2021-12-14T19:53:42.118101Z","shell.execute_reply.started":"2021-12-14T19:53:42.110645Z","shell.execute_reply":"2021-12-14T19:53:42.117117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cts = pd.read_csv('../input/jigsaw-toxic-severity-rating/comments_to_score.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-14T20:02:51.940408Z","iopub.execute_input":"2021-12-14T20:02:51.941212Z","iopub.status.idle":"2021-12-14T20:02:52.061969Z","shell.execute_reply.started":"2021-12-14T20:02:51.941174Z","shell.execute_reply":"2021-12-14T20:02:52.061034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cts","metadata":{"execution":{"iopub.status.busy":"2021-12-14T20:02:56.430996Z","iopub.execute_input":"2021-12-14T20:02:56.431814Z","iopub.status.idle":"2021-12-14T20:02:56.446233Z","shell.execute_reply.started":"2021-12-14T20:02:56.431773Z","shell.execute_reply":"2021-12-14T20:02:56.444988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}