{"cells":[{"metadata":{"_uuid":"d36a552974cd975871a9ba3d6913401767991ea7"},"cell_type":"markdown","source":"## I wanted to use some of Mithrillion's features\n- in concert with my custom features and with the featureset passed down via Chai-ta Tsai, Iprapas, and others\n- I kept crashing my kernel\n- So I needed to pare it down BEFORE bringing it into my main kernel"},{"metadata":{"trusted":true,"_uuid":"0c2a1f54e89787b101b8b7283e61f629419e279e"},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"10d720d6855438c5638e2b5409a136397a03948c"},"cell_type":"code","source":"#this list was arrived at in the following way:\n#ran cesium features + metadata through lgbm with trouble classes only\n#csFeatsToAdd=importances[importances['fold']==5].nlargest(40,'mean_gain').loc[:,'feature'].unique()\n#will try to remove redundancies\n#it turns out that 32 are not redundant (although possible collinearities where same basic feature has different name)\n\ncesiumFeaturesToConsider=['mjd_diff_det', 'distmod',\n       'flux_by_flux_ratio_sq__longest_strike_above_mean',\n       '__max_slope___2_', '__skew___4_',\n       'flux__longest_strike_above_mean',\n       '__median_absolute_deviation___2_', '__max_slope___3_',\n       '__freq_varrat___3_', '__percent_amplitude___3_',\n       '__percent_difference_flux_percentile___5_', '__std___5_',\n       '__percent_amplitude___5_', '__median_absolute_deviation___1_',\n       '__freq2_rel_phase2___2_', 'hostgal_photoz',\n       '__freq_y_offset___0_', 'hostgal_photoz_certain',\n       '__stetson_j___5_', '__freq_varrat___1_',\n       '__qso_log_chi2_qsonu___0_', '__amplitude___2_',\n       '__percent_difference_flux_percentile___2_', '__amplitude___0_',\n       '__freq_varrat___5_', '__skew___5_', '__freq_varrat___2_',\n       '__freq3_freq___2_', '__freq1_rel_phase4___5_',\n       'flux__mean_change', '__flux_percentile_ratio_mid80___5_',\n       '__percent_amplitude___2_', '__amplitude___5_',\n       '__median_absolute_deviation___5_', '__freq3_freq___3_',\n       '__qso_log_chi2_qsonu___5_', 'hostgal_photoz_err',\n       '__freq1_rel_phase3___5_', '__freq2_rel_phase2___4_',\n       '__freq2_rel_phase3___4_','object_id']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0c2b25a13e4be59c2cfa019613e59b2a18d3eab3"},"cell_type":"markdown","source":"## First challenge - column names don't match in training and test\n- Features were chosen based on CV importances on training data\n- Opening the whole file crashed kernel\n- Trying to open with above column names weren't recognized\n- It is easier to go from testName --> trainingName using method below"},{"metadata":{"trusted":true,"_uuid":"3a54883fa66be243d7c103c8210e3fd3f880cb5a"},"cell_type":"code","source":"def convertNames(testName):\n    trainName=\"\"\n    \n    lenTest=len(testName)\n    for charindex in range(lenTest):\n        char=testName[charindex]\n        #print(char)\n        if char in [')','(', ' ',',',\"'\"]:\n            trainName=trainName + '_'\n            #print('changed')\n        else:\n            trainName+=char\n            \n    return trainName\n\n            \ntestName=\"('percent_amplitude', 0)\"\ntrainName=convertNames(testName)\nprint(trainName)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"62114eea59c31250e17c586ba68a8b619017b29d"},"cell_type":"markdown","source":"## Open the first 10 rows just to get the test column names\n"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#for chunk in pd.read_csv(fn, chunksize=10)\nfn='../input/plasticc-features/single_output_test_ts_features.csv'\nfor chunk in pd.read_csv(fn, chunksize=10):\n    testCols=chunk.columns\n    break","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f37dd9f11e4465973ac5134ff2ec738bbdd1d321"},"cell_type":"markdown","source":"## Get the column list and a dictionary for translating to the training names"},{"metadata":{"trusted":true,"_uuid":"b60a4d72af5e72057210e4097beeedf9344fd007"},"cell_type":"code","source":"nameDict={}\ncolsToGrab=[]\nfor testCol in testCols:\n    trainName=convertNames(testCol)\n    if trainName in cesiumFeaturesToConsider:\n        colsToGrab.append(testCol)\n        nameDict[testCol]=trainName\n        \n#colsToGrab\n    ","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0b3ed6f41d4e707c37265d6163709655362bac78"},"cell_type":"markdown","source":"## Open the selected columns of the dataFrame"},{"metadata":{"trusted":true,"_uuid":"5063f0eb19fa06fda5db9fb3d599ca921d0e6b54"},"cell_type":"code","source":"testCesiumDf = pd.read_csv(fn, skipinitialspace=True, usecols=colsToGrab)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a686ff8a2de3c1ca05eb5058221ceb0a778a804e"},"cell_type":"code","source":"testCesiumDf.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"be51e8ddd1e37a0e0d3fd8f27d2adbce814c4d58"},"cell_type":"markdown","source":"## Rename the columns so they'll match the training data"},{"metadata":{"trusted":true,"_uuid":"83ac167b910f90fce4f398165864ea283a4d83df"},"cell_type":"code","source":"#df = df.rename(columns={'oldName1': 'newName1', 'oldName2': 'newName2'})\ntestCesiumDf = testCesiumDf.rename(columns=nameDict)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f3588eb53f0592a059316e7445c0ca996638ab1e"},"cell_type":"markdown","source":"## Save the file for pulling into featureMergingKernel"},{"metadata":{"trusted":true,"_uuid":"76c40a7f4ecefe106fde011e696ce599cf5c024c"},"cell_type":"code","source":"testCesiumDf.to_csv('reducedCesiumTestRevB.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}