{"cells":[{"metadata":{"_uuid":"9b3c3d8ed894478446b41744732669b6637169e7"},"cell_type":"markdown","source":"## This kernel expands on **[this work](https://www.kaggle.com/qianchao/smote-with-imbalance-data)**\n- in addition to handling multiclass, it creates a method that can be called repeatedly\n- which is important because [SMOTE should be done AFTER cross validation splits](https://www.marcoaltini.com/blog/dealing-with-imbalanced-data-undersampling-oversampling-and-proper-cross-validation)**"},{"metadata":{"_uuid":"1a7f46f72a74294002b0bae320175d40b946a52d"},"cell_type":"markdown","source":"## Call this method on each population split during cross validation\n- It is NOT appropriate to use it prior to cross validation\n- and there's no guarantee every class will be in every cross validated sample\n"},{"metadata":{"trusted":true,"_uuid":"9f1bb1f335b9516c68efe233a2b366778c07c106"},"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\nfrom sklearn.model_selection import train_test_split\nimport numpy as np # linear algebra\nimport pandas as pd\n\ndef smoteAdataset(Xig, yig, test_size=0.2, random_state=0):\n    \n    Xig_train, Xig_test, yig_train, yig_test = train_test_split(Xig, yig, test_size=test_size, random_state=random_state)\n    print(\"Number transactions X_train dataset: \", Xig_train.shape)\n    print(\"Number transactions y_train dataset: \", yig_train.shape)\n    print(\"Number transactions X_test dataset: \", Xig_test.shape)\n    print(\"Number transactions y_test dataset: \", yig_test.shape)\n\n    classes=[]\n    for i in np.unique(yig):\n        classes.append(i)\n        print(\"Before OverSampling, counts of label \" + str(i) + \": {}\".format(sum(yig_train==i)))\n        \n    sm=SMOTE(random_state=2)\n    Xig_train_res, yig_train_res = sm.fit_sample(Xig_train, yig_train.ravel())\n\n    print('After OverSampling, the shape of train_X: {}'.format(Xig_train_res.shape))\n    print('After OverSampling, the shape of train_y: {} \\n'.format(yig_train_res.shape))\n    \n    for eachClass in classes:\n        print(\"After OverSampling, counts of label \" + str(eachClass) + \": {}\".format(sum(yig_train_res==eachClass)))\n        \n    return Xig_train_res, yig_train_res, Xig_test, yig_test\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"#!pip install modin\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/assemblecustomtestsetrevb\"))\nprint(os.listdir(\"../input/something-different\"))\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\ntraindf=pd.read_csv(\"../input/something-different/newTrainFeatureOutputUnprocessed.csv\")\nif 'Unnamed: 0' in traindf.columns:\n    traindf=traindf.drop('Unnamed: 0', axis=1)\n#traindf = traindf.round(8)\ntraindf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"03ba39d9fc89a5f4c5915d58400160ba2f98c259"},"cell_type":"code","source":"testdf=pd.read_csv(\"../input/assemblecustomtestsetrevb/completeTestSetRevB.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dce6a84ddc1c4f07f2d98fb5da3a392d40cc7307"},"cell_type":"code","source":"if 'Unnamed: 0' in testdf.columns:\n    testdf=testdf.drop('Unnamed: 0', axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58b6e8fe8dc9df524ac586029ce5d5e1923ce961"},"cell_type":"code","source":"\nprint(traindf.shape)\n#traindf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7bea3e81cb7fc16cac153a4d6713c50760949c17"},"cell_type":"code","source":"print(testdf.shape)\ntestdf.columns","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"286d1259121d02956b39049a282d83027f247bf3"},"cell_type":"markdown","source":"## Let's do some feature engineering on these features\n- Looking at [this discussion](https://www.kaggle.com/c/PLAsTiCC-2018/discussion/71827) for guidance\n- But hoping to repeat for each subpopulation"},{"metadata":{"trusted":true,"_uuid":"dcef332807f2acbb9948cfda0ad381f7ee831bac"},"cell_type":"code","source":"def repeatPerBand(df, prefix):\n    \n    df.loc[:,prefix + 'frsq']=df.loc[:,prefix+'med']**2 / df.loc[:,prefix+'mfl']**2\n    df.loc[:,prefix + 'frsqxf']=df.loc[:,prefix+'frsq'] * df.loc[:,prefix+'med']\n    \n    return df\n\nprefixes=['le','lm','ll','he','hm','hl']\nfor prefix in prefixes:\n    \n    traindf=repeatPerBand(traindf, prefix)\n    print(traindf.shape)\n    \n#traindf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cccf56ae0601cd73c43da510872a6479133eecac"},"cell_type":"markdown","source":"## Repeat for test"},{"metadata":{"trusted":true,"_uuid":"df95513cf0c7ce6bcdc4dc1a509df1e2601d7c50"},"cell_type":"code","source":"for prefix in prefixes:\n    \n    testdf=repeatPerBand(testdf, prefix)\n    print(testdf.shape)\n    \n#testdf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"155ac495314e0101d3273c262d484801c8a1416e"},"cell_type":"markdown","source":"## Let's get rid of features that shouldn't be used in the model\n- these are features on the way to other features but not necessarily interesting on their own"},{"metadata":{"trusted":true,"_uuid":"96a56c2b039af7c9b2a1e86f96b349bc7c5f550c"},"cell_type":"code","source":"def removePerBand(df, prefixes=['le','lm','ll','he','hm','hl']):\n    for prefix in prefixes:\n        df=df.drop([prefix + 'mxd', prefix + 'mnd'], axis=1)\n        print(df.shape)\n    \n\n    return df\n\ntraindf=removePerBand(traindf)  \n#traindf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"064d480a4eba3bb78269344875ca5b74972bee3d"},"cell_type":"markdown","source":"## Repeat for test"},{"metadata":{"trusted":true,"_uuid":"c7ca574f5c711cfce6ac6293f5f3f48b73ff5dc2"},"cell_type":"code","source":"testdf=removePerBand(testdf)  \n#testdf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"754634af0d0c55b535ebfcea862a9fadbdcba5d0"},"cell_type":"markdown","source":"## Check for nan"},{"metadata":{"trusted":true,"_uuid":"1987c6b5885ba5b38775bc131a3918e8f3b92be3"},"cell_type":"code","source":"#distmod\ntraindf['distmod'] = traindf['distmod'].fillna(value=0)\n#traindf.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fc9d0f370b908afb6a0e66c877f60aab2770d4db"},"cell_type":"markdown","source":"## Repeat for test"},{"metadata":{"trusted":true,"_uuid":"c2c2ad2103405f69da730b45b9e730d3d732899a"},"cell_type":"code","source":"#for column in testdf.columns:\n#    print(column)\n#    joe = testdf[column].isna().sum()\n#    if joe>0:\n#        print(joe)\ntestdf['distmod'] = testdf['distmod'].fillna(value=0)\nprint(testdf['distmod'].isna().sum())\n#hostgal_specz\n#distmod","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"40f8f8653fdc41daf96d92c726f834f379f582e5"},"cell_type":"markdown","source":"## since hostgal_specz is missing for most test records, we'll get rid of it\n- in both sets since there's no point in training on it"},{"metadata":{"trusted":true,"_uuid":"0f28cbda8a58ed1b2657a2f2ed8d08cd3bcb5154"},"cell_type":"code","source":"traindf=traindf.drop('hostgal_specz', axis=1)\ntestdf=testdf.drop('hostgal_specz', axis=1)\nprint(traindf.shape)\nprint(testdf.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ebc293c9248276d50bd4519153dc74bb6d6b177e"},"cell_type":"markdown","source":"## Let's try to get the distmod analog"},{"metadata":{"trusted":true,"_uuid":"ea270ed985ecd9d578bb5a66acc5f929dd84e249"},"cell_type":"code","source":"def photoztodist(df) :\n    dfhpz=(data[\"hostgal_photoz\"])\n    return ((((((np.log(((dfhpz + (np.log(((dfhpz + (np.sqrt((np.log((np.maximum(((3.0)), (dfhpz * 2.0))))))))))))))) + (12.99870681762695312))) + (1.17613816261291504))) * (3.0))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0bc6a1588050858ecafce994db3ff4e0d0835905"},"cell_type":"code","source":"#from Helgi's comment on Scirpus' Photoz2Distance kernel\n#np.log(meta _train.hostgal _photoz)\n#https://www.kaggle.com/scirpus/photoz2distance\n#traindf['photozDist']=0\n#distmodnonzero=traindf.loc[:,'distmod']>0\n#traindf.loc[distmodnonzero,'photozDist']=#np.log(traindf.loc[distmodnonzero,'hostgal_photoz'])\n#traindf['photozDist'] = traindf['photozDist'].fillna(value=0)\n#print(traindf.shape)\n#testdf['photozDist']=np.log(testdf['hostgal_photoz'])\n#testdf['photozDist'] = testdf['photozDist'].fillna(value=0)\n#print(testdf.shape)\n#traindf.describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"6cf4174f630c1f03bdb5c31faf7dd3e574703e03"},"cell_type":"markdown","source":"## Save the object_id and then remove for modeling"},{"metadata":{"trusted":true,"_uuid":"1847ffb01c95372326933ed9e2c6a03bf908a17a"},"cell_type":"code","source":"print(traindf.shape)\nprint(testdf.shape)\n\n#trobjids=traindf.loc[:,'object_id']\n#print(trobjids.shape)\n\n#traindf=traindf.drop('object_id',axis=1)\n#print(traindf.shape)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f87134cd839911d8b6ab5fe36e88399a0292161"},"cell_type":"code","source":"#teobjids=testdf.loc[:,'object_id']\n#print(teobjids.shape)\n#testdf=testdf.drop('object_id',axis=1)\n#print(testdf.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd762ea660f6c377a5caf8e4292fbc03e1672522"},"cell_type":"code","source":"#trobjdf=pd.DataFrame(columns=['object_id'], data=trobjids)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"66dd18e5d2435bf90a7e8b178f34ce6e1f12626a"},"cell_type":"code","source":"#print(trobjdf.shape)\n#teobjdf=pd.DataFrame(columns=['object_id'], data=teobjids)\n#print(teobjdf.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ad294181709b33106acf24d33d151dafa1c6e49"},"cell_type":"markdown","source":"\n## I don't think we need to split the dataset into inter and extra galactic\n- It was pretty darn easy for us to see the feature to split on\n- I trust that any algorithm we use will find it equally easy"},{"metadata":{"_uuid":"2dfdb9ed11557e67a505eb24c465d503d7289519"},"cell_type":"markdown","source":"## Lets deal with extreme outliers\n- need to keep track so you can reverse this method"},{"metadata":{"trusted":true,"_uuid":"6509191b60101810c0ef8640fe1a58af4f483d02"},"cell_type":"code","source":"def convertAllToLogBase(df, excludeLogCols=['ra', 'decl', 'gal_l', 'gal_b', 'hostgal_photoz', 'target',\n                                            'hostgal_photoz_err', 'distmod', 'outlierScore', 'object_id']):\n    \n    for cindex in df.columns:\n        if (cindex not in excludeLogCols) & (len(df.loc[:,cindex].unique()) > 2):\n            #zeroFilter=df.loc[:,cindex]==0 (stays zero)\n            negFilter=df.loc[:,cindex]<0\n            posFilter=df.loc[:,cindex]>0\n            \n            df.loc[negFilter,cindex]=-1.0*np.log(-1.0*df.loc[negFilter,cindex])\n            df.loc[posFilter,cindex]=np.log(df.loc[posFilter,cindex])\n            #print(cindex)\n    return df\n\ntraindf=convertAllToLogBase(traindf)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb85ad8db6a87c2c8312d3cad334f5481620b4a2"},"cell_type":"code","source":"testdf=convertAllToLogBase(testdf)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60afe5944b0c163e29aec4226c96f2993452817c"},"cell_type":"code","source":"def reduceExtremeOutliers(df,targCol='target', logStdRng=8.0):\n    \n    #possibleLog=[]\n    #nonLog=[]\n    for cindex in df.columns:\n        \n        if (cindex==targCol) | (len(df.loc[:,cindex].unique())<=2) | (cindex=='object_id'):\n            \n            print('doing nothing for ' + str(cindex))\n            \n        else:\n            stdev=np.std(df.loc[:,cindex])\n            minVal=np.min(df.loc[:,cindex])\n            theRange=np.max(df.loc[:,cindex])-minVal\n            stdRange=theRange / stdev\n            if logStdRng < stdRange:\n                #df.loc[:,cindex]=df.loc[:,cindex]-minVal\n                #df.loc[:,cindex]=log(df.loc[:,cindex])\n                print('flagged ' + str(cindex) + ' for high variability')\n                print(stdRange)\n                #possibleLog.append(cindex)\n            \n                \n            newStdev=np.std(df.loc[:,cindex])\n            med=np.median(df.loc[:,cindex])\n            tooBig=med+newStdev*logStdRng/2\n            tooLittle=med-newStdev*logStdRng/2\n            tooBigFilter=df.loc[:,cindex]>tooBig\n            tooLittleFilter=df.loc[:,cindex]<tooLittle\n            df.loc[tooBigFilter,cindex]=tooBig\n            df.loc[tooLittleFilter,cindex]=tooLittle\n            \n    return df #, possibleLog, nonLog\n            \ntraindf=reduceExtremeOutliers(traindf)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"adbbb81bd706ab45398bd2696f38c657ce8a6406"},"cell_type":"code","source":"testdf=reduceExtremeOutliers(testdf)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2b401b966855cf53f05433109cf694c1eade86b0"},"cell_type":"markdown","source":"## Feature scale and eliminate features that contain no information"},{"metadata":{"trusted":true,"_uuid":"2f27dbc9ee63cb79855667a7eb0ce2bb78035099"},"cell_type":"code","source":"def featureScaleAllExcept(df, targCol='target'):\n    \n    for cindex in df.columns:\n        if '_TF' not in cindex:\n            if (cindex != targCol) & (cindex !='object_id'):\n                minval=np.min(df.loc[:,cindex])\n                theRange=np.max(df.loc[:,cindex])-minval\n                if theRange==0:\n                    #this feature contains no information\n                    df=df.drop(cindex,axis=1)\n                    print('dropped ' + str(cindex) + ' from dataFrame')\n\n                elif type(df[cindex])==bool:\n                    print(str(cindex) + ' is a boolean, doing nothing')\n\n                elif theRange==1:\n                    print('the range for ' + str(cindex) + ' is already 1, doing nothing')\n\n                else:\n                    #feature scale\n                    df.loc[:,cindex]=(df.loc[:,cindex]-minval)/theRange\n                \n    return df\n\nprint(traindf.shape)\ntraindf=featureScaleAllExcept(traindf)\nprint(traindf.shape)\nprint(traindf.loc[:,'target'].unique())\n\nprint(testdf.shape)\ntestdf=featureScaleAllExcept(testdf)\nprint(testdf.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8ffa546f9bb7b5f2a5387ddfb730c716c691c84d"},"cell_type":"code","source":"print(traindf.shape)\n\nprint(testdf.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3f657f9f34d824ebab7305810c063955586d798"},"cell_type":"code","source":"#trobjdf.to_csv('trainObjectIds.csv',sep='\\t', encoding='utf-8', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c5d5c5c97c3b70b42bac2245903ffe8a1598eaf"},"cell_type":"code","source":"traindf = traindf.round(8)\ntraindf.to_csv('traindfNormal.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f5eac5fce011191d4f0ed502101cc60f55acc2ba"},"cell_type":"code","source":"#teobjdf.to_csv('testObjectIds.csv',sep='\\t', encoding='utf-8', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"737a35b9ee3f300b081ba8b630a46bd16d1b80af"},"cell_type":"code","source":"testdf = testdf.round(8)\ntestdf.to_csv('testdfNormal.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e7782a6de883f2de4ecadf2b6fda38a13b59b8e6"},"cell_type":"markdown","source":"## We're ready to start training models\n- We have oversampling training sets for both inter-galactic and extra-galactic objects"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}