{"cells":[{"metadata":{"_uuid":"6063b5d0a0806b59c417426d4e54cec82ad8e00b"},"cell_type":"markdown","source":"## What does this kernel do?\n- Take as input a submission file (the submission in the input for this file score 1.052)\n- Load that submission, apply my class 99 logic, and save it as a new submission\n- Use this on your favorite submission(s) for a small score bump"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"86b80663252b6ac07223b9a3293a839f15eec283"},"cell_type":"markdown","source":"## Get the predictions data frame (pdf)"},{"metadata":{"trusted":true,"_uuid":"e4fd6b72292cdff4c214dadf011dd46d28520baf"},"cell_type":"code","source":"pdf=pd.read_csv('../input/forked-dart-w-ideas-from-kernels-added-smote/single_subm_0.648970_2018-11-21-09-03.csv')\nprint(pdf.shape)\npdf.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"68c301d01a3b2503a5d87cc37d278f5e5bec7f83"},"cell_type":"code","source":"pdf.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2d801af9b563d1a375ea3dbd258505f11259d5af"},"cell_type":"markdown","source":"## What is a class 99 object?\n- I look up in the sky and I see something I've never seen before\n- Which means it shouldn't fit well into any of the other classes\n- When the maximum probability for a known class is high, Class 99 should be low\n- When it doesn't match anything in my training set very well, Class 99 should be high"},{"metadata":{"_uuid":"23c601c53ad70bcddc06590336eb626d65155f17"},"cell_type":"markdown","source":""},{"metadata":{"trusted":true,"_uuid":"ea200f97c3fb81ecc13cb1b24cb03d77d0c259bc"},"cell_type":"code","source":"#from Scirpus discussion:\n\ndef GenUnknown(data):\n    return ((((((data[\"mymedian\"]) + (((data[\"mymean\"]) / 2.0)))/2.0)) + (((((1.0) - (((data[\"mymax\"]) * (((data[\"mymax\"]) * (data[\"mymax\"]))))))) / 2.0)))/2.0)\n\nfeats = ['class_6', 'class_15', 'class_16', 'class_42', 'class_52', 'class_53',\n         'class_62', 'class_64', 'class_65', 'class_67', 'class_88', 'class_90',\n         'class_92', 'class_95']\n\ny = pd.DataFrame()\ny['mymean'] = pdf[feats].mean(axis=1)\ny['mymedian'] = pdf[feats].median(axis=1)\ny['mymax'] = pdf[feats].max(axis=1)\n\npdf['class_99'] = GenUnknown(y)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f25023248dce8e24373184b4a2f28aed5157842b"},"cell_type":"code","source":"pdf.describe()\n#for cindex in newpdf.columns:\n#    print(cindex)\n#    print(type(newpdf.loc[0,cindex]))\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bffdc2b0e113ac4abb2934ab302010b482b22a16"},"cell_type":"code","source":"pdf.to_csv('improvedClass99forSmoteDart.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":""}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}