{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook refers to https://github.com/MLWave/Kaggle-Ensemble-Guide\nwhich is the code implementation of  [Kaggle Ensembling Guide](https://web.archive.org/web/20160304031055/http://mlwave.com/kaggle-ensembling-guide/) including three ensemble methods.","metadata":{}},{"cell_type":"code","source":"from __future__ import division\nfrom collections import defaultdict, Counter\nfrom glob import glob\nimport math\nimport re\n","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:53:14.359741Z","iopub.execute_input":"2022-08-23T09:53:14.360389Z","iopub.status.idle":"2022-08-23T09:53:14.366289Z","shell.execute_reply.started":"2022-08-23T09:53:14.360351Z","shell.execute_reply":"2022-08-23T09:53:14.364989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"glob_files =[x for x in glob('../input/*/*.csv') if 'amex-default-prediction' not in x]\nprint(glob_files)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:53:24.453805Z","iopub.execute_input":"2022-08-23T09:53:24.454792Z","iopub.status.idle":"2022-08-23T09:53:24.467146Z","shell.execute_reply.started":"2022-08-23T09:53:24.454747Z","shell.execute_reply":"2022-08-23T09:53:24.465821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Voting ensembles.\nloc_outfile = '/kaggle/working/vote.csv'\n\nweights_strategy = \"uniform\"\n\ndef kaggle_bag(glob_files, loc_outfile, method=\"average\", weights=\"uniform\"):\n  pattern = re.compile(r\"(.)*_[w|W](\\d*)_[.]*\")\n  if method == \"average\":\n    scores = defaultdict(list)\n  with open(loc_outfile,\"w\") as outfile:\n    #weight_list may be usefull using a different method\n    weight_list = [1]*len(glob_files)\n    for i, glob_file in enumerate( glob_files ):\n      print(\"parsing: {}\".format(glob_file))\n      if weights == \"weighted\":\n         weight = pattern.match(glob_file)\n         if weight and weight.group(2):\n            print(\"Using weight: {}\".format(weight.group(2)))\n            weight_list[i] = weight_list[i]*int(weight.group(2))\n         else:\n            print(\"Using weight: 1\")\n      # sort glob_file by first column, ignoring the first line\n      lines = open(glob_file).readlines()\n      lines = [lines[0]] + sorted(lines[1:])\n      for e, line in enumerate( lines ):\n        if i == 0 and e == 0:\n          outfile.write(line)\n        if e > 0:\n          row = line.strip().split(\",\")\n          for l in range(1,weight_list[i]+1):\n            scores[(e,row[0])].append(row[1])\n    for j,k in sorted(scores):\n      outfile.write(\"%s,%s\\n\"%(k,Counter(scores[(j,k)]).most_common(1)[0][0]))\n    print(\"wrote to {}\".format(loc_outfile))\n\nkaggle_bag(glob_files, loc_outfile, weights=weights_strategy)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:56:10.295140Z","iopub.execute_input":"2022-08-23T09:56:10.295657Z","iopub.status.idle":"2022-08-23T09:56:27.311745Z","shell.execute_reply.started":"2022-08-23T09:56:10.295605Z","shell.execute_reply":"2022-08-23T09:56:27.310572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Geometric averaging \nloc_outfile = '/kaggle/working/geomean.csv'\ndef kaggle_bag(glob_files, loc_outfile, method=\"average\", weights=\"uniform\"):\n  if method == \"average\":\n    scores = defaultdict(float)\n  with open(loc_outfile,\"w\") as outfile:\n    for i, glob_file in enumerate( glob_files ):\n      print(\"parsing: {}\".format(glob_file))\n      # sort glob_file by first column, ignoring the first line\n      lines = open(glob_file).readlines()\n      lines = [lines[0]] + sorted(lines[1:])\n      for e, line in enumerate( lines ):\n        if i == 0 and e == 0:\n          outfile.write(line)\n        if e > 0:\n          row = line.strip().split(\",\")\n          if scores[(e,row[0])] == 0:\n            scores[(e,row[0])] = 1\n          scores[(e,row[0])] *= float(row[1])\n    for j,k in sorted(scores):\n      outfile.write(\"%s,%f\\n\"%(k,math.pow(scores[(j,k)],1/(i+1))))\n    print(\"wrote to {}\".format(loc_outfile))\n\nkaggle_bag(glob_files, loc_outfile)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:58:33.163760Z","iopub.execute_input":"2022-08-23T09:58:33.164168Z","iopub.status.idle":"2022-08-23T09:58:42.857047Z","shell.execute_reply.started":"2022-08-23T09:58:33.164136Z","shell.execute_reply":"2022-08-23T09:58:42.855756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Rank averaging\nloc_outfile = '/kaggle/working/rank.csv'\ndef kaggle_bag(glob_files, loc_outfile):\n  with open(loc_outfile,\"w\") as outfile:\n    all_ranks = defaultdict(list)\n    for i, glob_file in enumerate( glob_files ):\n      file_ranks = []\n      print(\"parsing: {}\".format(glob_file))\n      # sort glob_file by first column, ignoring the first line\n      lines = open(glob_file).readlines()\n      lines = [lines[0]] + sorted(lines[1:])\n      for e, line in enumerate( lines ):\n        if e == 0 and i == 0:\n          outfile.write( line )\n        elif e > 0:\n          r = line.strip().split(\",\")\n          file_ranks.append( (float(r[1]), e, r[0]) )\n      for rank, item in enumerate( sorted(file_ranks) ):\n        all_ranks[(item[1],item[2])].append(rank)\n    average_ranks = []\n    for k in sorted(all_ranks):\n      average_ranks.append((sum(all_ranks[k])/len(all_ranks[k]),k))\n    ranked_ranks = []\n    for rank, k in enumerate(sorted(average_ranks)):\n      ranked_ranks.append((k[1][0],k[1][1],rank/(len(average_ranks)-1)))\n    for k in sorted(ranked_ranks):\n      outfile.write(\"%s,%s\\n\"%(k[1],k[2]))\n    print(\"wrote to {}\".format(loc_outfile))\n\nkaggle_bag(glob_files, loc_outfile)","metadata":{"execution":{"iopub.status.busy":"2022-08-23T09:59:14.478480Z","iopub.execute_input":"2022-08-23T09:59:14.478834Z","iopub.status.idle":"2022-08-23T09:59:44.384522Z","shell.execute_reply.started":"2022-08-23T09:59:14.478801Z","shell.execute_reply":"2022-08-23T09:59:44.383664Z"},"trusted":true},"execution_count":null,"outputs":[]}]}