{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport numpy as np\nimport pandas as pd\nsys.path.append('../input/handm-submission-files')\nfrom average_precision import apk\nfrom sklearn.model_selection import ParameterGrid\nfrom tqdm import tqdm_notebook as tqdm\nimport itertools\nimport random\nimport pickle\n\ndef pickle_dump(obj, path):\n    with open(path, mode='wb') as f:\n        pickle.dump(obj,f)\n\ndef pickle_load(path):\n    with open(path, mode='rb') as f:\n        data = pickle.load(f)\n        return data\n    \ndef stringToListInt(string):\n    listRes = list(string.split(\" \"))\n    listRes = [int(s) for s in listRes]\n    return listRes\n\ndef stringToList(string):\n    listRes = list(string.split(\" \"))\n    listRes = [int(s) for s in listRes]\n    return listRes\n\ndef get_oof_score(answer_df, oof_df):\n    apks = []\n    for answer, pred in zip(answer_df['answer'].values, oof_df['prediction'].values):\n        answer = stringToList(answer)\n        pred = stringToListInt(pred)\n        apks.append(apk(answer, pred[:12]))\n    score = np.mean(apks)\n    #print(score)\n    return score\n\ndef read_csvs_and_output_predict_columns(pahtes):\n    validation_length = 68984\n    predict_columns = []\n    for i, path in enumerate(pathes):\n        if i == 0:\n            master_oof = pd.read_csv(path).sort_values('customer_id').reset_index(drop=True)[['customer_id', 'prediction']]\n            master_oof.columns = [f'customer_id_sub{i}', f'prediction{i}']\n            predict_columns.append(f'prediction{i}')\n        else:\n            sub = pd.read_csv(path).sort_values('customer_id').reset_index(drop=True)[['customer_id', 'prediction']]\n            sub.columns = [f'customer_id_sub{i}', f'prediction{i}']\n            predict_columns.append(f'prediction{i}')\n            master_oof = pd.concat([master_oof, sub], axis=1)\n\n            bool_cutomer_same = (master_oof['customer_id_sub0'] == master_oof[f'customer_id_sub{i}']).sum() == validation_length\n            print('check_length')\n            if bool_cutomer_same:\n                print('OK')\n            else:\n                print('NG')\n            master_oof = master_oof.drop([f'customer_id_sub{i}'], axis=1)\n    return master_oof, predict_columns\n\ndef cust_blend(dt, predict_columns = [],W = [1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n    REC = []\n    for predict_column in predict_columns:\n        REC.append(dt[predict_column].split())\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])\n\ndef get_REC(dt, predict_columns = []):\n    REC = []\n    for predict_column in predict_columns:\n        REC.append(dt[predict_column].split())\n    return REC\n\ndef cust_blend_custom(REC, predict_columns = [],W = [1,1,1]):\n    #Global ensemble weights\n    #W = [1.15,0.95,0.85]\n    \n    #Create a list of all model predictions\n#    REC = []\n#   for predict_column in predict_columns:\n#        REC.append(dt[predict_column].split())\n    \n    #Create a dictionary of items recommended. \n    #Assign a weight according the order of appearance and multiply by global weights\n    res = {}\n    for M in range(len(REC)):\n        for n, v in enumerate(REC[M]):\n            if v in res:\n                res[v] += (W[M]/(n+1))\n            else:\n                res[v] = (W[M]/(n+1))\n    \n    # Sort dictionary by item weights\n    res = list(dict(sorted(res.items(), key=lambda item: -item[1])).keys())\n    \n    # Return the top 12 itens only\n    return ' '.join(res[:12])","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-08T12:13:52.042747Z","iopub.execute_input":"2022-05-08T12:13:52.043108Z","iopub.status.idle":"2022-05-08T12:13:53.113599Z","shell.execute_reply.started":"2022-05-08T12:13:52.043065Z","shell.execute_reply":"2022-05-08T12:13:53.112864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"answer_df = pd.read_csv('../input/handm-submission-files/validation_answer.csv').sort_values('customer_id').reset_index(drop=True)[['customer_id', 'answer']]\nanswer_customers = answer_df['customer_id'].to_list()","metadata":{"execution":{"iopub.status.busy":"2022-05-08T12:14:04.397239Z","iopub.execute_input":"2022-05-08T12:14:04.397630Z","iopub.status.idle":"2022-05-08T12:14:04.680837Z","shell.execute_reply.started":"2022-05-08T12:14:04.397602Z","shell.execute_reply":"2022-05-08T12:14:04.679870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fix hiroi san oof\npathes = [\n    '../input/handm-submission-files/byfone_Chris_agetop12_repeatuser_validation.csv', \n    '../input/handm-submission-files/exp13_LightSANs_week2_4_8_ensemble_validation.csv',\n]\n\ndef fix_customers_get_fixed_pathes(answer_df, pathes):\n    fixed_pathes = []\n    for path in pathes:\n        _df = pd.read_csv(path)\n        mask = _df['customer_id'].isin(answer_customers)\n        _df = _df[mask].reset_index(drop=True)\n        _df.to_csv(os.path.basename(path), index=False)\n        fixed_pathes.append(os.path.basename(path))\n    return fixed_pathes\n\nfixed_pathes = fix_customers_get_fixed_pathes(answer_df, pathes)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T12:14:05.826433Z","iopub.execute_input":"2022-05-08T12:14:05.826692Z","iopub.status.idle":"2022-05-08T12:14:18.961475Z","shell.execute_reply.started":"2022-05-08T12:14:05.826666Z","shell.execute_reply":"2022-05-08T12:14:18.960625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pathes = [\n    '../input/fork-of-h-m-eda-rule-base-by-customer-age-make-oof/submission_validation.csv', \n    '../input/handm-submission-files/exp038_validation_pred.csv',\n    '../input/handm-submission-files/h-m-pure-pytorch-baseline_exp010_submission_oof.csv'\n]\npathes.extend(fixed_pathes)\n\nparams = list(itertools.product(\n    np.linspace(0.2, 1, 5).tolist(), \n    np.linspace(0.2, 1, 5).tolist(),\n    np.linspace(0.2, 1, 5).tolist(),\n    np.linspace(0.5, 1.5, 5).tolist(),\n    np.linspace(0.5, 1.5, 5).tolist()\n    ))\n\nrandom.seed(0)\nrandom.shuffle(params)\n\nmaster_oof, predict_columns = read_csvs_and_output_predict_columns(pathes)\n\nmaster_oof['prediction'] = master_oof.apply(cust_blend, predict_columns = predict_columns, W = [0, 0, 1, 0, 0], axis=1)\n\nget_oof_score(answer_df, master_oof)","metadata":{"execution":{"iopub.status.busy":"2022-05-08T12:14:18.963031Z","iopub.execute_input":"2022-05-08T12:14:18.963243Z","iopub.status.idle":"2022-05-08T12:14:26.533062Z","shell.execute_reply.started":"2022-05-08T12:14:18.963218Z","shell.execute_reply":"2022-05-08T12:14:26.532493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nbest_score = 0\n\nfor param in tqdm(params):\n    param = list(param)\n    master_oof['prediction'] = master_oof.apply(cust_blend, predict_columns = predict_columns, W = param, axis=1)\n\n    score = get_oof_score(answer_df, master_oof)\n    \n    if best_score < score:\n        print(f'best_score improve:{score:.6f}_param{param}')\n        best_score = score\n        best_param = param\nprint('finish!!')\nprint(f'best_score was {best_score} param {best_param}')","metadata":{"execution":{"iopub.status.busy":"2022-05-08T12:14:37.820178Z","iopub.execute_input":"2022-05-08T12:14:37.820478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}