{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":10684,"databundleVersionId":230682,"sourceType":"competition"}],"dockerImageVersionId":30513,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# importing libraries\n\nimport pandas as pd, numpy as np, matplotlib.pyplot as plt, seaborn as sns\nimport pyarrow.parquet as pq\nimport os\nfrom keras.layers import *\nfrom keras.models import Model\nfrom tqdm import tqdm\nfrom sklearn.metrics import accuracy_score \nfrom sklearn.model_selection import train_test_split\nfrom keras import backend as K\nfrom keras import optimizers\n\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom keras.callbacks import *\nfrom keras import initializers, regularizers, constraints","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-20T05:02:15.928072Z","iopub.execute_input":"2023-06-20T05:02:15.928494Z","iopub.status.idle":"2023-06-20T05:02:15.935678Z","shell.execute_reply.started":"2023-06-20T05:02:15.928459Z","shell.execute_reply":"2023-06-20T05:02:15.934906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many foldswil be created\nN_SPLITS = 5\n\n# it is just a constant with the measurement data size\nsample_size = 800000\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:12:30.719111Z","iopub.execute_input":"2023-06-20T07:12:30.720962Z","iopub.status.idle":"2023-06-20T07:12:30.728614Z","shell.execute_reply.started":"2023-06-20T07:12:30.720903Z","shell.execute_reply":"2023-06-20T07:12:30.727142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef matthews_correlation(y_true, y_pred):\n    '''\n        Calclualtes the Matthews correlation coefficient measure for quality of\n        binary classification probled\n    '''\n\n    y_pred_pos = K.round(K.clip(y_pred, 0, 1))\n    y_pred_neg = 1 - y_pred_pos\n    \n    y_pos = K.round(K.clip(y_true, 0, 1))\n    y_neg = 1 - y_pos\n    \n    \n    tp = K.sum(y_pos * y_pred_pos)\n    tn = K.sum(y_neg * y_pred_neg)\n    \n    \n    fp = K.sum(y_neg * y_pred_pos)\n    fn = K.sum(y_pos * y_pred_neg)\n    \n    \n    numerator = (tp * tn  - fp * fn)\n    \n    denominator = K.sqrt((tp + fp) * (tp + fn) * (tn + fp) * (tn + fn))\n\n    return numerator / (denominator + K.epsilon())\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-20T03:42:57.608558Z","iopub.execute_input":"2023-06-20T03:42:57.609305Z","iopub.status.idle":"2023-06-20T03:42:57.617588Z","shell.execute_reply.started":"2023-06-20T03:42:57.609263Z","shell.execute_reply":"2023-06-20T03:42:57.616517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass Attention(Layer):\n    def __init__(self, step_dim,\n                 W_regularizer=None, b_regularizer=None,\n                 W_constraint=None, b_constraint=None,\n                 bias=True, **kwargs):\n        self.supports_masking = True\n        self.init = initializers.get('glorot_uniform')\n\n        self.W_regularizer = regularizers.get(W_regularizer)\n        self.b_regularizer = regularizers.get(b_regularizer)\n\n        self.W_constraint = constraints.get(W_constraint)\n        self.b_constraint = constraints.get(b_constraint)\n\n        self.bias = bias\n        self.step_dim = step_dim\n        self.features_dim = 0\n        super(Attention, self).__init__(**kwargs)\n\n    def build(self, input_shape):\n        assert len(input_shape) == 3\n\n        self.W = self.add_weight(shape =  (input_shape[-1],),\n                                 initializer=self.init,\n                                 name='{}_W'.format(self.name),\n                                 regularizer=self.W_regularizer,\n                                 constraint=self.W_constraint)\n        self.features_dim = input_shape[-1]\n\n        if self.bias:\n            self.b = self.add_weight(shape = (input_shape[1],),\n                                     initializer='zero',\n                                     name='{}_b'.format(self.name),\n                                     regularizer=self.b_regularizer,\n                                     constraint=self.b_constraint)\n        else:\n            self.b = None\n\n        self.built = True\n\n    def compute_mask(self, input, input_mask=None):\n        return None\n\n    def call(self, x, mask=None):\n        features_dim = self.features_dim\n        step_dim = self.step_dim\n\n        eij = K.reshape(K.dot(K.reshape(x, (-1, features_dim)),\n                        K.reshape(self.W, (features_dim, 1))), (-1, step_dim))\n\n        if self.bias:\n            eij += self.b\n\n        eij = K.tanh(eij)\n\n        a = K.exp(eij)\n\n        if mask is not None:\n            a *= K.cast(mask, K.floatx())\n\n        a /= K.cast(K.sum(a, axis=1, keepdims=True) + K.epsilon(), K.floatx())\n\n        a = K.expand_dims(a)\n        weighted_input = x * a\n        return K.sum(weighted_input, axis=1)\n\n    def compute_output_shape(self, input_shape):\n        return input_shape[0],  self.features_dim","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:41:30.149007Z","iopub.execute_input":"2023-06-20T12:41:30.149366Z","iopub.status.idle":"2023-06-20T12:41:30.166278Z","shell.execute_reply.started":"2023-06-20T12:41:30.149337Z","shell.execute_reply":"2023-06-20T12:41:30.165349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Loading","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/vsb-power-line-fault-detection/metadata_train.csv')\ndf_train = df_train.set_index(['id_measurement', 'phase'])\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T05:13:52.829812Z","iopub.execute_input":"2023-06-20T05:13:52.830191Z","iopub.status.idle":"2023-06-20T05:13:52.898132Z","shell.execute_reply.started":"2023-06-20T05:13:52.830163Z","shell.execute_reply":"2023-06-20T05:13:52.897201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_num = 127\nmin_num = -128\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T05:56:30.108820Z","iopub.execute_input":"2023-06-20T05:56:30.109268Z","iopub.status.idle":"2023-06-20T05:56:30.114650Z","shell.execute_reply.started":"2023-06-20T05:56:30.109218Z","shell.execute_reply":"2023-06-20T05:56:30.113428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function standardize the data rom ( -128 to 127 ) to ( -1 to 1 )\n## So this prevents onve vvalue to dominate others \n\n\ndef min_max_transf(ts, min_data, max_data,range_needed=(-1,1)):\n\n    if min_data < 0:\n\n        ts_std = (ts + abs(min_data)) / (max_data + abs(min_data))\n    \n    else:\n        \n        ts_std = (ts - min_data) / (max_data - min_data)\n\n    \n    if range_needed[0] < 0:\n        return ts_std * ( range_needed[1] + abs(range_needed[0]) ) + range_needed[0]\n    else:\n        return ts_std * (range_needed[1] - range_needed[0]) + range_needed[0]\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T06:02:07.395561Z","iopub.execute_input":"2023-06-20T06:02:07.395958Z","iopub.status.idle":"2023-06-20T06:02:07.405105Z","shell.execute_reply.started":"2023-06-20T06:02:07.395927Z","shell.execute_reply":"2023-06-20T06:02:07.403791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\ndef transform_ts(ts,n_dim = 160, min_max = (-1,1)):\n    # Convert data to -1 to 1\n    ts_std = min_max_transf(ts, min_data=min_num, max_data = max_num)\n    # bucket or chunck size, 5000 in this case\n    \n    bucket_size = int( sample_size / n_dim)\n\n    new_ts = []\n    \n    for i in range(0, sample_size, bucket_size):\n        # cut each bucket to ts_range\n\n        ts_range = ts_std[i:i+ bucket_size]\n        \n        # Calculate each feature\n        mean = ts_range.mean()\n        std = ts_range.std()\n        std_top = mean + std\n        \n        std_bot = mean - std\n        \n        percentile_calc = np.percentile(\n            ts_range, \n            [0,1,25,50,75,99,100]\n        )\n        \n        max_range = percentile_calc[-1] - percentile_calc[0]\n        \n        relative_percentile = percentile_calc - mean\n\n        new_ts.append(\n            np.concatenate(\n                [ \n                    np.asarray(\n                        [\n                            mean,\n                            std,\n                            std_top,\n                            std_bot,\n                            max_range\n                        ]\n                    )\n                    ,percentile_calc,relative_percentile\n                ]\n            ),\n            \n        )\n        \n    return np.asarray(new_ts)\n        \n        \n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:14:40.441399Z","iopub.execute_input":"2023-06-20T07:14:40.441870Z","iopub.status.idle":"2023-06-20T07:14:40.451232Z","shell.execute_reply.started":"2023-06-20T07:14:40.441836Z","shell.execute_reply":"2023-06-20T07:14:40.450217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:23:21.742949Z","iopub.execute_input":"2023-06-20T12:23:21.743440Z","iopub.status.idle":"2023-06-20T12:23:21.748167Z","shell.execute_reply.started":"2023-06-20T12:23:21.743371Z","shell.execute_reply":"2023-06-20T12:23:21.747080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.loc[5].loc[1]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:02:57.513309Z","iopub.execute_input":"2023-06-20T12:02:57.514707Z","iopub.status.idle":"2023-06-20T12:02:57.524393Z","shell.execute_reply.started":"2023-06-20T12:02:57.514650Z","shell.execute_reply":"2023-06-20T12:02:57.523215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Now comes a functiont which is going to take a peice of daata and convert using transofrm_ts\n## but it does to each of the 3 phases\n## If we would try to do in one time, could exceed the RAM Memmory\n\ndef prep_data(start,end):\n    praq_train = pq.read_pandas(\n        \"/kaggle/input/vsb-power-line-fault-detection/train.parquet\",\n        columns = [ str(i) for i in range(start,end)]).to_pandas()\n\n    X= []\n    Y= []\n\n    # Using tqdm to evaluate processing time\n    # Takes each index from df_train and iteract it from start to end\n    \n    for id_measurement in tqdm(\n            df_train.index.levels[0].unique()[int(start/3):int(end/3)]\n    ):\n        X_signal = []\n        # for each phase of sinal\n        \n        for phase in [0,1,2]:\n            # Extract from df_train both signal_id and target\n            signal_id, target = df_train.loc[id_measurement].loc[phase]\n            # But just append the target one time, to not triplicate it\n            if phase == 0:\n                Y.append(target)\n            \n            X_signal.append(transform_ts(praq_train[str(signal_id)]))\n        ## Concatinate all thre phase in one matrix\n        X_signal = np.concatenate(X_signal, axis = 1)\n        # add the data to X\n        X.append(X_signal)\n    X = np.asarray(X)\n    Y = np.asarray(Y)\n    return X, Y        \n            \n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:10:57.034102Z","iopub.execute_input":"2023-06-20T07:10:57.034528Z","iopub.status.idle":"2023-06-20T07:10:57.044909Z","shell.execute_reply.started":"2023-06-20T07:10:57.034495Z","shell.execute_reply":"2023-06-20T07:10:57.043706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nX = []\nY = []\n\n\ndef load_all():\n    total_size = len(df_train)\n    \n    for ini, end in [\n        (0,int(total_size / 2)),\n        (int(total_size / 2),total_size)\n    ]:\n        X_temp, Y_temp = prep_data(ini, end)\n        X.append(X_temp)\n        Y.append(Y_temp)\n    \n\nload_all()\n\n    \n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:14:49.001994Z","iopub.execute_input":"2023-06-20T07:14:49.002362Z","iopub.status.idle":"2023-06-20T07:34:50.966797Z","shell.execute_reply.started":"2023-06-20T07:14:49.002335Z","shell.execute_reply":"2023-06-20T07:34:50.965740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:35:33.898472Z","iopub.execute_input":"2023-06-20T07:35:33.899298Z","iopub.status.idle":"2023-06-20T07:35:33.905440Z","shell.execute_reply.started":"2023-06-20T07:35:33.899244Z","shell.execute_reply":"2023-06-20T07:35:33.904756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.concatenate(X)\nY= np.concatenate(Y)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:35:53.978630Z","iopub.execute_input":"2023-06-20T07:35:53.979034Z","iopub.status.idle":"2023-06-20T07:35:54.077969Z","shell.execute_reply.started":"2023-06-20T07:35:53.979005Z","shell.execute_reply":"2023-06-20T07:35:54.076805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:27:50.424963Z","iopub.execute_input":"2023-06-20T12:27:50.425372Z","iopub.status.idle":"2023-06-20T12:27:50.432137Z","shell.execute_reply.started":"2023-06-20T12:27:50.425339Z","shell.execute_reply":"2023-06-20T12:27:50.430637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Y.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:36:19.102119Z","iopub.execute_input":"2023-06-20T07:36:19.102495Z","iopub.status.idle":"2023-06-20T07:36:19.110130Z","shell.execute_reply.started":"2023-06-20T07:36:19.102466Z","shell.execute_reply":"2023-06-20T07:36:19.108898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"set(Y)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:37:22.598580Z","iopub.execute_input":"2023-06-20T07:37:22.599447Z","iopub.status.idle":"2023-06-20T07:37:22.606879Z","shell.execute_reply.started":"2023-06-20T07:37:22.599409Z","shell.execute_reply":"2023-06-20T07:37:22.605646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape, Y.shape)\n# save data into file, a numpy specific format\nnp.save(\"X.npy\",X)\nnp.save(\"y.npy\",Y)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T07:38:53.754957Z","iopub.execute_input":"2023-06-20T07:38:53.755336Z","iopub.status.idle":"2023-06-20T07:38:54.065842Z","shell.execute_reply.started":"2023-06-20T07:38:53.755307Z","shell.execute_reply":"2023-06-20T07:38:54.064869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"def model_lstm(input_shape):\n    # The shape was explained above, must have this order\n    inp = Input(shape=(input_shape[1], input_shape[2],))\n    # This is the LSTM layer\n    # Bidirecional implies that the 160 chunks are calculated in both ways, 0 to 159 and 159 to zero\n    # although it appear that just 0 to 159 way matter, I have tested with and without, and tha later worked best\n    # 128 and 64 are the number of cells used, too many can overfit and too few can underfit\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(inp)\n    # The second LSTM can give more fire power to the model, but can overfit it too\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    # Attention is a new tecnology that can be applyed to a Recurrent NN to give more meanings to a signal found in the middle\n    # of the data, it helps more in longs chains of data. A normal RNN give all the responsibility of detect the signal\n    # to the last cell. Google RNN Attention for more information :)\n    x = Attention(input_shape[1])(x)\n    # A intermediate full connected (Dense) can help to deal with nonlinears outputs\n    x = Dense(64, activation=\"relu\")(x)\n    # A binnary classification as this must finish with shape (1,)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    # Pay attention in the addition of matthews_correlation metric in the compilation, it is a success factor key\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=[matthews_correlation])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:34:36.793730Z","iopub.execute_input":"2023-06-20T12:34:36.794099Z","iopub.status.idle":"2023-06-20T12:34:36.802978Z","shell.execute_reply.started":"2023-06-20T12:34:36.794071Z","shell.execute_reply":"2023-06-20T12:34:36.800550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Traianing  KFold","metadata":{}},{"cell_type":"code","source":"# First, create a set of indexes of the 5 folds\nsplits = list(StratifiedKFold(n_splits=N_SPLITS, shuffle=True, random_state=2019).split(X, Y))\npreds_val = []\ny_val = []\n# Then, iteract with each fold\n# If you dont know, enumerate(['a', 'b', 'c']) returns [(0, 'a'), (1, 'b'), (2, 'c')]\nfor idx, (train_idx, val_idx) in enumerate(splits):\n    K.clear_session() # I dont know what it do, but I imagine that it \"clear session\" :)\n    print(\"Beginning fold {}\".format(idx+1))\n    # use the indexes to extract the folds in the train and validation data\n    train_X, train_y, val_X, val_y = X[train_idx], Y[train_idx], X[val_idx], Y[val_idx]\n    # instantiate the model for this fold\n    print(train_X.shape[1])\n    model = model_lstm(train_X.shape)\n    # This checkpoint helps to avoid overfitting. It just save the weights of the model if it delivered an\n    # validation matthews_correlation greater than the last one.\n    ckpt = ModelCheckpoint('weights_{}.h5'.format(idx), save_best_only=True, save_weights_only=True, verbose=1, monitor='val_matthews_correlation', mode='max')\n    # Train, train, train\n    model.fit(train_X, train_y, batch_size=128, epochs=50, validation_data=[val_X, val_y], callbacks=[ckpt])\n    # loads the best weights saved by the checkpoint\n    model.load_weights('weights_{}.h5'.format(idx))\n    # Add the predictions of the validation to the list preds_val\n    preds_val.append(model.predict(val_X, batch_size=512))\n    # and the val true y\n    y_val.append(val_y)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:41:38.821689Z","iopub.execute_input":"2023-06-20T12:41:38.822362Z","iopub.status.idle":"2023-06-20T12:47:59.072431Z","shell.execute_reply.started":"2023-06-20T12:41:38.822328Z","shell.execute_reply":"2023-06-20T12:47:59.071432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Now output of kernal must be binary but otuput of nn in 0 to 1 in sigmoid range\n\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in tqdm([i * 0.01 for i in range(100)]):\n        score = K.eval(matthews_correlation(y_true, (y_proba > threshold)))\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    search_result = {'threshold': best_threshold, 'matthews_correlation': best_score}\n    return search_result","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:04:13.268110Z","iopub.execute_input":"2023-06-20T13:04:13.268482Z","iopub.status.idle":"2023-06-20T13:04:13.275039Z","shell.execute_reply.started":"2023-06-20T13:04:13.268451Z","shell.execute_reply":"2023-06-20T13:04:13.273901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_threshold = threshold_search(y_val, preds_val)['threshold']","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:04:16.914539Z","iopub.execute_input":"2023-06-20T13:04:16.915435Z","iopub.status.idle":"2023-06-20T13:04:16.982292Z","shell.execute_reply.started":"2023-06-20T13:04:16.915398Z","shell.execute_reply":"2023-06-20T13:04:16.980895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Now load the test data\n# This first part is the meta data, not the main data, the measurements\nmeta_test = pd.read_csv('/kaggle/input/vsb-power-line-fault-detection/metadata_test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:07:38.338377Z","iopub.execute_input":"2023-06-20T13:07:38.338760Z","iopub.status.idle":"2023-06-20T13:07:38.375653Z","shell.execute_reply.started":"2023-06-20T13:07:38.338729Z","shell.execute_reply":"2023-06-20T13:07:38.374390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_test = meta_test.set_index(['signal_id'])\nmeta_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:07:44.551409Z","iopub.execute_input":"2023-06-20T13:07:44.551793Z","iopub.status.idle":"2023-06-20T13:07:44.569190Z","shell.execute_reply.started":"2023-06-20T13:07:44.551761Z","shell.execute_reply":"2023-06-20T13:07:44.568315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# First we daclarete a series of parameters to initiate the loading of the main data\n# it is too large, it is impossible to load in one time, so we are doing it in dividing in 10 parts\nfirst_sig = meta_test.index[0]\nn_parts = 10\nmax_line = len(meta_test)\npart_size = int(max_line / n_parts)\nlast_part = max_line % n_parts\nprint(first_sig, n_parts, max_line, part_size, last_part, n_parts * part_size + last_part)\n# Here we create a list of lists with start index and end index for each of the 10 parts and one for the last partial part\nstart_end = [[x, x+part_size] for x in range(first_sig, max_line + first_sig, part_size)]\nstart_end = start_end[:-1] + [[start_end[-1][0], start_end[-1][0] + last_part]]\nprint(start_end)\nX_test = []\n# now, very like we did above with the train data, we convert the test data part by part\n# transforming the 3 phases 800000 measurement in matrix (160,57)\nfor start, end in start_end:\n    subset_test = pq.read_pandas('/kaggle/input/vsb-power-line-fault-detection/test.parquet', columns=[str(i) for i in range(start, end)]).to_pandas()\n    for i in tqdm(subset_test.columns):\n        id_measurement, phase = meta_test.loc[int(i)]\n        subset_test_col = subset_test[i]\n        subset_trans = transform_ts(subset_test_col)\n        X_test.append([i, id_measurement, phase, subset_trans])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}