{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n##\nprint(os.listdir(\"../input/neuralmoredecisive\"))\nprint(os.listdir(\"../input/adjustlgbm1003\"))\nprint(os.listdir(\"../input/mergesvm\"))\n\n##\nprint(os.listdir(\"../input/ourpathtowherewearenewparams\"))\nprint(os.listdir(\"../input/adjustpredictiondataframe\"))\n\n###\nprint(os.listdir(\"../input/closethegapbetweencvandlb\"))\nprint(os.listdir(\"../input/betternormalizedwithxgb\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"215a42f7fa5b06f334b3ab531b2badb3c1e6b0d2"},"cell_type":"code","source":"##\n#df1010 = pd.read_csv(\"../input/ourpathtowherewearenewparams/subm_0.610175_2018-12-07-23-29.csv\")\n#df1010=df1010.sort_values('object_id').reset_index(drop=True)\n#df992 = pd.read_csv(\"../input/adjustpredictiondataframe/justSetZeroProbas.csv\")\n#df992=df992.sort_values('object_id').reset_index(drop=True)\n\n##\n#dftestize= pd.read_csv(\"../input/closethegapbetweencvandlb/subm_0.668470_2018-12-06-19-58.csv\")\n#dftestize=dftestize.sort_values('object_id').reset_index(drop=True)\n#dfxgb= pd.read_csv(\"../input/betternormalizedwithxgb/subm_0.644318_2018-12-11-18-25.csv\")\n#dfxgb=dfxgb.sort_values('object_id').reset_index(drop=True)\n\n#print(df1010.shape)\n#print(df992.shape)\n#print(dftestize.shape)\n#print(dfxgb.shape)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"## Fundamentally different models:\n#dfsvm = pd.read_csv(\"../input/mergesvm/mergeOriginalSvm.csv\") #I have never submitted this on its own\n#dfsvm=dfsvm.sort_values('object_id').reset_index(drop=True)\ndflgbm = pd.read_csv(\"../input/adjustlgbm1003/justSetZeroProbas.csv\") #haven't scored this alone, similar 1.024\ndflgbm=dflgbm.sort_values('object_id').reset_index(drop=True)\ndfnn=pd.read_csv(\"../input/neuralmoredecisive/moreDecisiveNeural.csv\")\ndfnn=dfnn.sort_values('object_id').reset_index(drop=True)\n\n#print(dfsvm.shape)\nprint(dflgbm.shape)\nprint(dfnn.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42f3a919c599133f057f95858c290eb721de4054"},"cell_type":"code","source":"dflgbm.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d3c576b3ed25716e7da3f20485c883ec1013fcc"},"cell_type":"code","source":"dfBlend=.5*dflgbm + .5*dfnn #+ .1*(df992 + df1010 + dfxgb)\ndfBlend.loc[:,'object_id']=dflgbm.loc[:,'object_id']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5ddd60fc2ccac4c079d2162b703d7fd8ea1e8d1"},"cell_type":"code","source":"dfBlend.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9f7e93114cb86a1e1f42f0eb5a058303dd15e3ad"},"cell_type":"code","source":"dfBlend.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91bde514266a3485ab1da0420777b5d7b49d9960"},"cell_type":"code","source":"dfBlend.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3b8556960f40854b6a1f9d060f4b91cabce5d852"},"cell_type":"code","source":"import random\n#dfBlend=.35*dflgbm + .25*dfnn + .1*(df992 + df1010 + dftestize + dfxgb)\ndfs=[dflgbm, dfnn]\nwts=[.5, .5]\n#num per class\nnpc=1000\n\n\ndef validateBlend(bdf, dfs, wts, npc=1000):\n    \n    recs=bdf.shape[0]\n    cols=len(bdf.columns)\n    failures=0\n    for i in range(npc):\n        record=random.randint(0,recs-1)\n        feat=bdf.columns[random.randint(0, cols-1)]\n        \n        \n        shouldBe=0\n        isVal=bdf.loc[record, feat]\n        for j in range(len(wts)):\n            \n            shouldBe += dfs[j].loc[record, feat] * wts[j]\n        shouldBe=round(shouldBe,6)\n        isVal=round(isVal,6)\n        if shouldBe==isVal:\n            #print('match')\n            dosomething=False\n        else:\n            print(record)\n            print(feat)\n            print('should be: ' + str(shouldBe))\n            print('is: ' + str(isVal))\n            failures+=1\n            \n    print('total failures: ' + str(failures) + ' out of ' + str(npc) +' sample points')\n    return failures\n            \n    \nfailures=validateBlend(dfBlend, dfs, wts)    \n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d15583b7d58fa408ac2dd0fee9a0b7d60ffcaea"},"cell_type":"code","source":"dfBlend.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"242f90449188ad52292efb47100d7c5096741228"},"cell_type":"code","source":"dfBlend.to_csv('lgbm50DecisiveNeural50.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}