{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Context\n\nIn this notebook, I try to test how a good (stability) score can be made by using fixed predictions for each type of discussion.\n\n```\nTYPES = train_origin['discourse_type'].unique()\n```\n\nTo do this, I randomly generated predictions and tried to find the optimal combination of them.\n\n```\n>>> get_preds()\n[0.34, 0.44, 0.22]\n```\n\n```\nrandom_preds = {type_name: get_preds(n_round=1) for type_name in TYPES}\n```\n\n\nI am spliting the data into training (to search for values) and validation.\n\n```\ntrain_indx, val_indx = train_test_split(\n    df.index, stratify=df['SplitBy'].values,\n    test_size=0.15, random_state=2022\n)\n\ncols_list = ['Type', 'Label', 'Target']\ntrain_data = df.loc[train_indx, cols_list].copy().reset_index(drop=True)\nvalid_data = df.loc[val_indx, cols_list].copy().reset_index(drop=True)\n```\n\nI tried different options - some of them were more stable, and others were less.\n\n```\nhow_split = 1\n\nif how_split == 1:\n    df['SplitBy'] = df['Type'] + '-' + df['Label']\n    \nelif how_split == 2:\n    df['SplitBy'] = df['Type']\n    \nelif how_split == 3:\n    df['SplitBy'] = train_origin['essay_id']\n    \nelse:\n    df['SplitBy'] = df['Label']\n``` \n\n### Statistics (SplitBy =)\n\n- essay_id\n\nScore | Valid | Diff\n-- | -- | --\n1.012 | 1.007 | +0,005\n1.018 | 1.018 | 0\n1.025 | 1.021 | +0,004\n1.032 | 1.023 | +0,009\n1.036 | 1.034 | +0,002\n\n- type / discourse_type\n\nScore | Valid | Diff\n-- | -- | --\n1.003 | 1.005 | -0,002\n1.025 | 1.028 | -0,003\n1.027 | 1.029 | -0,002\n1.029 | 1.027 | +0,002\n1.032 | 1.034 | -0,002\n\n- label / discourse_effectiveness\n\nScore | Valid | Diff\n-- | -- | --\n1.001 | 1.005 | -0,004\n1.025 | 1.023 | +0,002\n1.034 | 1.035 | -0,001\n1.036 | 1.033 | +0,003\n1.038 | 1.028 | +0,010\n\n- type-label\n\nScore | Valid | Diff\n-- | -- | --\n1.022 | 1.022 | 0\n1.025 | 1.025 | 0\n1.027 | 1.027 | 0\n1.028 | 1.028 | 0\n1.029 | 1.028 | +0,001","metadata":{}},{"cell_type":"markdown","source":"# 1. Import & Def & Set & Load","metadata":{}},{"cell_type":"code","source":"import random\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import train_test_split","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.045525,"end_time":"2022-07-11T17:24:29.588372","exception":false,"start_time":"2022-07-11T17:24:28.542847","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T00:08:22.348666Z","iopub.execute_input":"2022-07-12T00:08:22.349165Z","iopub.status.idle":"2022-07-12T00:08:23.592561Z","shell.execute_reply.started":"2022-07-12T00:08:22.349064Z","shell.execute_reply":"2022-07-12T00:08:23.591462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_count_essay(essay_id, data):\n    return data.isin([essay_id]).sum()\n\n\ndef get_preds(n_preds=3, n_round=2):\n    result = []\n    value = 1\n    \n    for _ in range(n_preds-1): \n        if value > 0:\n            x = round(random.uniform(0, value), n_round)\n            value = round(value - x, n_round)\n        else:\n            x = 0\n            \n        result.append(x)\n    \n    result.append(value)\n        \n    return result\n\ndef fill_preds(data, mapping):\n    predictions = []\n    for x in data:\n        predictions.append(mapping.get(x))\n        \n    return np.array(predictions)\n\n\ndef get_score(y_true, predictions, n_round=3):\n    result = log_loss(y_true, predictions)\n    \n    return round(result, n_round)","metadata":{"papermill":{"duration":0.018364,"end_time":"2022-07-11T17:24:29.611400","exception":false,"start_time":"2022-07-11T17:24:29.593036","status":"completed"},"tags":[],"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T00:08:23.596983Z","iopub.execute_input":"2022-07-12T00:08:23.597343Z","iopub.status.idle":"2022-07-12T00:08:23.606920Z","shell.execute_reply.started":"2022-07-12T00:08:23.597311Z","shell.execute_reply":"2022-07-12T00:08:23.606104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_path = \"../input/feedback-prize-effectiveness/train.csv\"\ncols_list = ['essay_id', 'discourse_text', 'discourse_type', 'discourse_effectiveness']\n\ntrain_origin = pd.read_csv(data_path, usecols=cols_list)\ntrain_origin.head()","metadata":{"papermill":{"duration":0.419833,"end_time":"2022-07-11T17:24:30.035807","exception":false,"start_time":"2022-07-11T17:24:29.615974","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:08:23.608077Z","iopub.execute_input":"2022-07-12T00:08:23.608539Z","iopub.status.idle":"2022-07-12T00:08:23.927189Z","shell.execute_reply.started":"2022-07-12T00:08:23.608509Z","shell.execute_reply":"2022-07-12T00:08:23.926228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TYPES = train_origin['discourse_type'].unique()\nprint(*TYPES)","metadata":{"papermill":{"duration":0.021707,"end_time":"2022-07-11T17:24:30.062066","exception":false,"start_time":"2022-07-11T17:24:30.040359","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:08:23.929420Z","iopub.execute_input":"2022-07-12T00:08:23.929977Z","iopub.status.idle":"2022-07-12T00:08:23.944652Z","shell.execute_reply.started":"2022-07-12T00:08:23.929910Z","shell.execute_reply":"2022-07-12T00:08:23.943922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LABELS = ['Ineffective', 'Adequate', 'Effective']\n\ndf = pd.DataFrame({\n    'Type': train_origin['discourse_type'],\n    'Label': train_origin['discourse_effectiveness']\n})\n\nmapping = {name: i for i, name in enumerate(LABELS)}\ndf['Target'] = df['Label'].replace(mapping)\n\nhow_split = 1\n\nif how_split == 1:\n    df['SplitBy'] = df['Type'] + '-' + df['Label']\n    \nelif how_split == 2:\n    df['SplitBy'] = df['Type']\n    \nelif how_split == 3:\n    df['SplitBy'] = train_origin['essay_id']\n    \n    # We can't use train_test_split\n    # if essay_id has only one value.\n    count_essay = train_origin['essay_id'].transform(\n        get_count_essay, data=train_origin['essay_id']\n    )\n    df = df.loc[count_essay > 1, :]\n    print('> 1:', (count_essay > 1).sum())  # 36689\n    print('= 1:', (count_essay == 1).sum()) # 76\n    \nelse:\n    df['SplitBy'] = df['Label']\n\ndf.head(10)","metadata":{"papermill":{"duration":0.378066,"end_time":"2022-07-11T17:24:30.444667","exception":false,"start_time":"2022-07-11T17:24:30.066601","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:08:23.945892Z","iopub.execute_input":"2022-07-12T00:08:23.946365Z","iopub.status.idle":"2022-07-12T00:09:21.221992Z","shell.execute_reply.started":"2022-07-12T00:08:23.946336Z","shell.execute_reply":"2022-07-12T00:09:21.220709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. Check simply fixed predictions","metadata":{}},{"cell_type":"code","source":"fixed_preds = df.loc[: , ['Label', 'Target']]\n\n_ = [.1, .7, .2]\nfixed_preds[LABELS] = _\n\nfixed_preds.head()","metadata":{"papermill":{"duration":0.032338,"end_time":"2022-07-11T17:24:30.481657","exception":false,"start_time":"2022-07-11T17:24:30.449319","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.223696Z","iopub.execute_input":"2022-07-12T00:09:21.224187Z","iopub.status.idle":"2022-07-12T00:09:21.246460Z","shell.execute_reply.started":"2022-07-12T00:09:21.224138Z","shell.execute_reply":"2022-07-12T00:09:21.244964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"get_score(fixed_preds['Target'], fixed_preds[LABELS])","metadata":{"papermill":{"duration":0.030497,"end_time":"2022-07-11T17:24:30.516978","exception":false,"start_time":"2022-07-11T17:24:30.486481","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.248055Z","iopub.execute_input":"2022-07-12T00:09:21.249251Z","iopub.status.idle":"2022-07-12T00:09:21.275830Z","shell.execute_reply.started":"2022-07-12T00:09:21.249213Z","shell.execute_reply":"2022-07-12T00:09:21.273838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. Split train data","metadata":{}},{"cell_type":"code","source":"train_indx, val_indx = train_test_split(\n    df.index, stratify=df['SplitBy'].values,\n    test_size=0.15, random_state=2022\n)","metadata":{"papermill":{"duration":0.046053,"end_time":"2022-07-11T17:24:30.568517","exception":false,"start_time":"2022-07-11T17:24:30.522464","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.277211Z","iopub.execute_input":"2022-07-12T00:09:21.278088Z","iopub.status.idle":"2022-07-12T00:09:21.413696Z","shell.execute_reply.started":"2022-07-12T00:09:21.278054Z","shell.execute_reply":"2022-07-12T00:09:21.412490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_list = ['Type', 'Label', 'Target']\ntrain_data = df.loc[train_indx, cols_list].copy().reset_index(drop=True)\nvalid_data = df.loc[val_indx, cols_list].copy().reset_index(drop=True)\n\nprint(train_data.shape)\nprint(valid_data.shape)","metadata":{"papermill":{"duration":0.022887,"end_time":"2022-07-11T17:24:30.596402","exception":false,"start_time":"2022-07-11T17:24:30.573515","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.417149Z","iopub.execute_input":"2022-07-12T00:09:21.417507Z","iopub.status.idle":"2022-07-12T00:09:21.433370Z","shell.execute_reply.started":"2022-07-12T00:09:21.417475Z","shell.execute_reply":"2022-07-12T00:09:21.432246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"papermill":{"duration":0.0198,"end_time":"2022-07-11T17:24:30.621334","exception":false,"start_time":"2022-07-11T17:24:30.601534","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.434878Z","iopub.execute_input":"2022-07-12T00:09:21.435243Z","iopub.status.idle":"2022-07-12T00:09:21.448388Z","shell.execute_reply.started":"2022-07-12T00:09:21.435211Z","shell.execute_reply":"2022-07-12T00:09:21.447208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. Check mapping (random predictions)","metadata":{}},{"cell_type":"code","source":"get_preds()","metadata":{"papermill":{"duration":0.013638,"end_time":"2022-07-11T17:24:30.640178","exception":false,"start_time":"2022-07-11T17:24:30.626540","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.449859Z","iopub.execute_input":"2022-07-12T00:09:21.450646Z","iopub.status.idle":"2022-07-12T00:09:21.463461Z","shell.execute_reply.started":"2022-07-12T00:09:21.450597Z","shell.execute_reply":"2022-07-12T00:09:21.462336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_mapping = {type_name: get_preds() for type_name in TYPES}\n\nfor key, values in random_mapping.items():\n    print(key, values)","metadata":{"papermill":{"duration":0.015254,"end_time":"2022-07-11T17:24:30.660617","exception":false,"start_time":"2022-07-11T17:24:30.645363","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.465044Z","iopub.execute_input":"2022-07-12T00:09:21.466093Z","iopub.status.idle":"2022-07-12T00:09:21.477000Z","shell.execute_reply.started":"2022-07-12T00:09:21.466043Z","shell.execute_reply":"2022-07-12T00:09:21.475836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_data = df.loc[:10, 'Type']\n\nfill_preds(check_data, random_mapping)","metadata":{"papermill":{"duration":0.014439,"end_time":"2022-07-11T17:24:30.680516","exception":false,"start_time":"2022-07-11T17:24:30.666077","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.478682Z","iopub.execute_input":"2022-07-12T00:09:21.479433Z","iopub.status.idle":"2022-07-12T00:09:21.492647Z","shell.execute_reply.started":"2022-07-12T00:09:21.479386Z","shell.execute_reply":"2022-07-12T00:09:21.491286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.DataFrame(\n    fill_preds(check_data, random_mapping),\n    columns=LABELS\n)","metadata":{"papermill":{"duration":0.020197,"end_time":"2022-07-11T17:24:30.705963","exception":false,"start_time":"2022-07-11T17:24:30.685766","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.494548Z","iopub.execute_input":"2022-07-12T00:09:21.495468Z","iopub.status.idle":"2022-07-12T00:09:21.514824Z","shell.execute_reply.started":"2022-07-12T00:09:21.495420Z","shell.execute_reply":"2022-07-12T00:09:21.513646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. Train loop (get best_scores)","metadata":{}},{"cell_type":"code","source":"steps = 450000  # 300000 / 600000\ncheckpoints = [10000, 100000, 150000, 200000, 250000, 300000, 400000, 500000]\n\nprint(f'steps: \\t{steps}')","metadata":{"papermill":{"duration":0.016422,"end_time":"2022-07-11T17:24:30.728221","exception":false,"start_time":"2022-07-11T17:24:30.711799","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.516662Z","iopub.execute_input":"2022-07-12T00:09:21.517264Z","iopub.status.idle":"2022-07-12T00:09:21.528686Z","shell.execute_reply.started":"2022-07-12T00:09:21.517220Z","shell.execute_reply":"2022-07-12T00:09:21.527175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntext_col = \"Type\"\ntarget_col = \"Target\"\nx_train = train_data[text_col]\ny_train = train_data[target_col]\n\nbest_scores = {}\n\nfor step in range(1, steps):\n    random_preds = {type_name: get_preds(n_round=1) for type_name in TYPES}\n    \n    predictions = fill_preds(x_train, random_preds)\n    score = get_score(y_train, predictions)\n    \n    if score < 1.05:\n        best_scores[score] = random_preds\n        \n    if step in checkpoints:\n        best_score = sorted(best_scores.keys())\n        checkpoint = f\"checkpoint: {step:<8}\"\n        if best_score:\n            first_score = best_score[0]\n            predictions = fill_preds(\n                valid_data[text_col], best_scores.get(first_score)\n            )\n            valid_score = get_score(\n                valid_data[target_col], predictions\n            )\n            len_scores = len(best_scores.keys())\n            print(f'{checkpoint} | {first_score:^5} | {valid_score:^5} |  {len_scores}')\n        else:\n            print(checkpoint)\n            \nprint('\\nThe end...\\n')","metadata":{"papermill":{"duration":11068.394008,"end_time":"2022-07-11T20:28:59.128008","exception":false,"start_time":"2022-07-11T17:24:30.734000","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:09:21.531170Z","iopub.execute_input":"2022-07-12T00:09:21.532188Z","iopub.status.idle":"2022-07-12T00:10:57.019345Z","shell.execute_reply.started":"2022-07-12T00:09:21.532135Z","shell.execute_reply":"2022-07-12T00:10:57.018194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. Display scores / predictions","metadata":{}},{"cell_type":"code","source":"print(len(best_scores.keys()))\nprint(sorted(best_scores.keys())[:10])","metadata":{"papermill":{"duration":0.10876,"end_time":"2022-07-11T20:28:59.244842","exception":false,"start_time":"2022-07-11T20:28:59.136082","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:10:57.021112Z","iopub.execute_input":"2022-07-12T00:10:57.021591Z","iopub.status.idle":"2022-07-12T00:10:57.029135Z","shell.execute_reply.started":"2022-07-12T00:10:57.021515Z","shell.execute_reply":"2022-07-12T00:10:57.027718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_best = 15\nscores_list = sorted(best_scores.keys())[:n_best]\n\nfor x in scores_list:\n    x_data = best_scores.get(x)\n    x_df = pd.DataFrame.from_dict(x_data, orient='index',\n                                  columns=LABELS)\n    \n    predictions = fill_preds(\n        valid_data[text_col], x_data\n    )\n    valid_score = get_score(\n        valid_data[target_col], predictions\n    )\n    \n    print(f'*** Score: {x} / Valid: {valid_score} ***')\n    display(x_df)","metadata":{"papermill":{"duration":0.257171,"end_time":"2022-07-11T20:28:59.508931","exception":false,"start_time":"2022-07-11T20:28:59.251760","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-12T00:10:57.030659Z","iopub.execute_input":"2022-07-12T00:10:57.031113Z","iopub.status.idle":"2022-07-12T00:10:57.337166Z","shell.execute_reply.started":"2022-07-12T00:10:57.031076Z","shell.execute_reply":"2022-07-12T00:10:57.335797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.010126,"end_time":"2022-07-11T20:28:59.530388","exception":false,"start_time":"2022-07-11T20:28:59.520262","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}