{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Is this experiment reproducible?\n\nIf it fails for a certain measurement at a certain loading, I would expect it to fail when I increase the loading for the same measurement.\n\nUsing KdTree, collect close data and check for reproducibility","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.preprocessing import StandardScaler\nfrom scipy.spatial import KDTree","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:42:41.397709Z","iopub.execute_input":"2022-08-03T03:42:41.398506Z","iopub.status.idle":"2022-08-03T03:42:41.702990Z","shell.execute_reply.started":"2022-08-03T03:42:41.398463Z","shell.execute_reply":"2022-08-03T03:42:41.701658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.set_option('display.max_columns', 200)\npd.set_option('display.max_rows', 200)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:40:08.839804Z","iopub.execute_input":"2022-08-03T03:40:08.840235Z","iopub.status.idle":"2022-08-03T03:40:08.847585Z","shell.execute_reply.started":"2022-08-03T03:40:08.840202Z","shell.execute_reply":"2022-08-03T03:40:08.845975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/tabular-playground-series-aug-2022/train.csv', index_col='id')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:26:10.346383Z","iopub.execute_input":"2022-08-03T03:26:10.346876Z","iopub.status.idle":"2022-08-03T03:26:10.465301Z","shell.execute_reply.started":"2022-08-03T03:26:10.346839Z","shell.execute_reply":"2022-08-03T03:26:10.464059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(train.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:26:12.658957Z","iopub.execute_input":"2022-08-03T03:26:12.659477Z","iopub.status.idle":"2022-08-03T03:26:13.002232Z","shell.execute_reply.started":"2022-08-03T03:26:12.659436Z","shell.execute_reply":"2022-08-03T03:26:13.001077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"measurement_list = []\nfor col in train.columns:\n    if 'measurement' in col:\n        measurement_list.append(col)\nmeasurement_list","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:26:15.739403Z","iopub.execute_input":"2022-08-03T03:26:15.740245Z","iopub.status.idle":"2022-08-03T03:26:15.748968Z","shell.execute_reply.started":"2022-08-03T03:26:15.740202Z","shell.execute_reply":"2022-08-03T03:26:15.747495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scaling measures\nfor measurement in measurement_list:\n    scaler = StandardScaler()\n    train[measurement] = scaler.fit_transform(train[[measurement]])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:43:31.311264Z","iopub.execute_input":"2022-08-03T03:43:31.311694Z","iopub.status.idle":"2022-08-03T03:43:31.400300Z","shell.execute_reply.started":"2022-08-03T03:43:31.311660Z","shell.execute_reply":"2022-08-03T03:43:31.399376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:43:43.538757Z","iopub.execute_input":"2022-08-03T03:43:43.539148Z","iopub.status.idle":"2022-08-03T03:43:43.647459Z","shell.execute_reply.started":"2022-08-03T03:43:43.539117Z","shell.execute_reply":"2022-08-03T03:43:43.646268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kd_data = train[measurement_list]\nkdtree = KDTree(kd_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:43:50.053421Z","iopub.execute_input":"2022-08-03T03:43:50.053802Z","iopub.status.idle":"2022-08-03T03:43:50.078749Z","shell.execute_reply.started":"2022-08-03T03:43:50.053772Z","shell.execute_reply":"2022-08-03T03:43:50.077603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"kd_data.iloc[1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:43:52.216975Z","iopub.execute_input":"2022-08-03T03:43:52.218219Z","iopub.status.idle":"2022-08-03T03:43:52.226801Z","shell.execute_reply.started":"2022-08-03T03:43:52.218159Z","shell.execute_reply":"2022-08-03T03:43:52.225819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's output loading, failure, id, dist in this order\n\nIt seems that if it fails once, it does not necessarily fail after increasing the load. Is this true?","metadata":{}},{"cell_type":"code","source":"# If you want to save a lot, please change the range\nfor row in range(0, 3):\n    # get neightbors\n    print(\"=\"*10, row, \"=\"*10)\n    nbh_d_list, nbh_row_list = kdtree.query(kd_data.iloc[row], k=20)\n    load_fail_list = []\n    for nbh_d, nbh_row in zip(nbh_d_list, nbh_row_list):\n        failuer = train[\"failure\"][nbh_row]\n        loading = train[\"loading\"][nbh_row]\n        load_fail_list.append((int(loading), failuer, nbh_row, nbh_d))\n    load_fail_list.sort()\n    print(*load_fail_list, sep=\"\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:50:13.392881Z","iopub.execute_input":"2022-08-03T03:50:13.393347Z","iopub.status.idle":"2022-08-03T03:50:13.418809Z","shell.execute_reply.started":"2022-08-03T03:50:13.393312Z","shell.execute_reply":"2022-08-03T03:50:13.417872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To be sure, let's look at the data that was determined to be close.\n\nEach measurement has a standard deviation of 1 due to scaling.\nWhen you think about it, there are many data that are off by one standard deviation level, which means that the records are not as similar as you might think.","metadata":{}},{"cell_type":"code","source":"row = 0\nnbh_d_list, nbh_row_list = kdtree.query(kd_data.iloc[row], k=20)\ntrain.iloc[nbh_row_list]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T03:44:15.172325Z","iopub.execute_input":"2022-08-03T03:44:15.173017Z","iopub.status.idle":"2022-08-03T03:44:15.226669Z","shell.execute_reply.started":"2022-08-03T03:44:15.172982Z","shell.execute_reply":"2022-08-03T03:44:15.225061Z"},"trusted":true},"execution_count":null,"outputs":[]}]}