{"cells":[{"metadata":{"_uuid":"41d767001ba114a834e3e83936322a16f50150ee"},"cell_type":"markdown","source":"Hey everyone!\nI am very new on Kaggle Competitions, and here I want tho share a different solution that I tried (with a bad LB score and quite inefficient to run) but perhaps may give someone a new idea or whatever. The goal is to search each test segment pattern in the train set, and assign the 'time_to_failure' value in that way."},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#Import dependencies\nimport numpy as np\nimport pandas as pd\nimport cv2\nimport matplotlib.pyplot as plt\nimport gc\nimport glob\nimport os\nfrom tqdm import tqdm","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8580cb646b939ab0739746db9203d4b6d88e1c3"},"cell_type":"code","source":"sub_file = pd.read_csv('../input/sample_submission.csv', index_col=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5316848d95a59b61d217e7b7615f863b09e7ba6a"},"cell_type":"code","source":"%%time\n#Load data from train file\ntrain = pd.read_csv('../input/train.csv', dtype={'acoustic_data': np.float32, 'time_to_failure': np.float32})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df6acf5c7d2db05ad56f135c646626c101380a27"},"cell_type":"code","source":"#Split the big train series into a fixed number of segments. Searching directly in the full\n#train series was too much for my computer \nnumber_train_segments = 3\ntrain_segment_length = int(train.shape[0] / number_train_segments)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1f603b7cebec4e87e3b73e9b74eed5f7b7bd27e"},"cell_type":"code","source":"#Read segment test data names\ntests = glob.glob('../input/test/**')\ntests_names = os.listdir('../input/test/')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b1b0ca9afe7122cd059cfc0ce0ac01580e173e6"},"cell_type":"code","source":"#Search the test pattern in the train data segments using the correlation coefficient and OpenCV\nsub_list = []\nfor j in tqdm(range(5)): #range(len(tests))): Only 5 as an example, because commiting with all of them takes too much\n                         #I only want to share the methodology\n    #Read segment data\n    segment_test = pd.read_csv(tests[j], dtype={'acoustic_data': np.float32})\n    segment_test = np.float32( segment_test['acoustic_data'].values )\n    #Resize the vector to have the correct dimensions\n    segment_test_tp = np.resize(segment_test, (1,len(segment_test)))\n\n    coefs = []\n    for i in range(0,number_train_segments):\n        print('Searching similarity for test segment {} with {}-segment of train data:'.format(tests_names[j], i+1))\n        segment_train = train.iloc[train_segment_length*i : train_segment_length*(i+1)]\n        segment_train = np.float32( segment_train['acoustic_data'].values )\n        segment_train_tp = np.resize( segment_train, (1,len(segment_train)) )\n    \n        gc.collect()\n    \n        result = cv2.matchTemplate(segment_test_tp, segment_train_tp, cv2.TM_CCORR_NORMED)\n        \n        #Append the best matching for that train segment (coeff and position)\n        coefs.append([np.max(result), train_segment_length*i + np.argmax(result) + segment_test_tp.shape[1]-1])\n    \n    #Apprend the best result among all train segments\n    coefs = np.array(coefs)\n    sub_list.append( [ tests_names[j], train.time_to_failure.iloc[int(coefs[np.argmax(coefs[:,0]),1])] ] )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76a6705b1edcedcabbcf77f567b4e69d20a4cc7d"},"cell_type":"code","source":"sub_df = pd.DataFrame(data=sub_list, columns=['seg_id','time_to_failure'])\nsub_df['seg_id'] = sub_df['seg_id'].apply(lambda x: x[:-4])\nsub_df.set_index('seg_id',inplace=True)\n\n#Read submission_file and rearrange the index in the \nsub_file = pd.read_csv('../input/sample_submission.csv', index_col=0)\nsub_df = sub_df.reindex(sub_file.index)\nsub_file = sub_df\n\nsub_file.to_csv('sub_file_v00.csv')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e1b35bf09e1f48ca4f6bbf17b74a0988cf128828"},"cell_type":"markdown","source":"Any suggestions, thoughts, ideas, complains, and feedback are welcome!"}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}