{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":87793,"databundleVersionId":11553390,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:40.917880Z","iopub.execute_input":"2025-04-12T10:13:40.918381Z","iopub.status.idle":"2025-04-12T10:13:41.405887Z","shell.execute_reply.started":"2025-04-12T10:13:40.918318Z","shell.execute_reply":"2025-04-12T10:13:41.404534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_labels.csv')\ntrain_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/train_sequences.csv')\nsubmission = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/sample_submission.csv')\ntest_sequences = pd.read_csv('/kaggle/input/stanford-rna-3d-folding/test_sequences.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.407354Z","iopub.execute_input":"2025-04-12T10:13:41.408009Z","iopub.status.idle":"2025-04-12T10:13:41.700758Z","shell.execute_reply.started":"2025-04-12T10:13:41.407978Z","shell.execute_reply":"2025-04-12T10:13:41.699594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.702584Z","iopub.execute_input":"2025-04-12T10:13:41.702879Z","iopub.status.idle":"2025-04-12T10:13:41.720861Z","shell.execute_reply.started":"2025-04-12T10:13:41.702855Z","shell.execute_reply":"2025-04-12T10:13:41.719661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_labels.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.722741Z","iopub.execute_input":"2025-04-12T10:13:41.723185Z","iopub.status.idle":"2025-04-12T10:13:41.739958Z","shell.execute_reply.started":"2025-04-12T10:13:41.723136Z","shell.execute_reply":"2025-04-12T10:13:41.738595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_sequences.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.741054Z","iopub.execute_input":"2025-04-12T10:13:41.741506Z","iopub.status.idle":"2025-04-12T10:13:41.769258Z","shell.execute_reply.started":"2025-04-12T10:13:41.741465Z","shell.execute_reply":"2025-04-12T10:13:41.768087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_sequences.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.770428Z","iopub.execute_input":"2025-04-12T10:13:41.770805Z","iopub.status.idle":"2025-04-12T10:13:41.788242Z","shell.execute_reply.started":"2025-04-12T10:13:41.770769Z","shell.execute_reply":"2025-04-12T10:13:41.786926Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## left join ##\ntrain_labels['ID'] = train_labels['ID'].str.rsplit('_', n=1).str[0]\ntrain_df  = train_labels.merge(how = 'left' , left_on = 'ID' , right_on = 'target_id' , right = train_sequences )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:41.789083Z","iopub.execute_input":"2025-04-12T10:13:41.789426Z","iopub.status.idle":"2025-04-12T10:13:42.085394Z","shell.execute_reply.started":"2025-04-12T10:13:41.789354Z","shell.execute_reply":"2025-04-12T10:13:42.084322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('No of rows and cols ' , train_df.shape)\nprint()\nprint('Missing values' , train_df.isna().sum())\nprint()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.087898Z","iopub.execute_input":"2025-04-12T10:13:42.088173Z","iopub.status.idle":"2025-04-12T10:13:42.151312Z","shell.execute_reply.started":"2025-04-12T10:13:42.088149Z","shell.execute_reply":"2025-04-12T10:13:42.150187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##datetime conversion ###\ntrain_df['temporal_cutoff'] = pd.to_datetime(train_df['temporal_cutoff']).astype('int64') // 10**9\ntest_sequences['temporal_cutoff'] = pd.to_datetime(test_sequences['temporal_cutoff']).astype('int64') // 10**9\n## using len of sequence \ntrain_df['seq_length'] = train_df['sequence'].str.len()\ntest_sequences['seq_length'] =  test_sequences['sequence'].str.len()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.152429Z","iopub.execute_input":"2025-04-12T10:13:42.152680Z","iopub.status.idle":"2025-04-12T10:13:42.227451Z","shell.execute_reply.started":"2025-04-12T10:13:42.152658Z","shell.execute_reply":"2025-04-12T10:13:42.226553Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID']  = submission['ID'].str.rsplit('_' ,n =1  ).str[0]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.228338Z","iopub.execute_input":"2025-04-12T10:13:42.228631Z","iopub.status.idle":"2025-04-12T10:13:42.237405Z","shell.execute_reply.started":"2025-04-12T10:13:42.228607Z","shell.execute_reply":"2025-04-12T10:13:42.235989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test  = test_sequences.merge(how = 'left' , left_on = 'target_id' , right_on = 'ID' , right = submission)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.238617Z","iopub.execute_input":"2025-04-12T10:13:42.238989Z","iopub.status.idle":"2025-04-12T10:13:42.261237Z","shell.execute_reply.started":"2025-04-12T10:13:42.238946Z","shell.execute_reply":"2025-04-12T10:13:42.260310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define columns\ncategorical_cols = ['resname', 'target_id', 'description', 'all_sequences']\nnumerical_cols = ['temporal_cutoff', 'resid', 'seq_length']\ntarget_cols = ['x_1', 'y_1', 'z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.262166Z","iopub.execute_input":"2025-04-12T10:13:42.262540Z","iopub.status.idle":"2025-04-12T10:13:42.279647Z","shell.execute_reply.started":"2025-04-12T10:13:42.262505Z","shell.execute_reply":"2025-04-12T10:13:42.278535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Encode categoricals\nfrom sklearn.preprocessing import LabelEncoder\n\nlabel_encoders = {}\n\nfor col in categorical_cols:\n    \n    le = LabelEncoder()\n    train_df[col] = train_df[col].astype(str).fillna('missing')\n    train_df[col] = le.fit_transform(train_df[col])\n    label_encoders[col] = le\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.280883Z","iopub.execute_input":"2025-04-12T10:13:42.281273Z","iopub.status.idle":"2025-04-12T10:13:42.962804Z","shell.execute_reply.started":"2025-04-12T10:13:42.281206Z","shell.execute_reply":"2025-04-12T10:13:42.961622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Get vocab sizes\nvocab_sizes = {col : train_df[col].nunique() + 1  for col in categorical_cols}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.964807Z","iopub.execute_input":"2025-04-12T10:13:42.965399Z","iopub.status.idle":"2025-04-12T10:13:42.975546Z","shell.execute_reply.started":"2025-04-12T10:13:42.965351Z","shell.execute_reply":"2025-04-12T10:13:42.974463Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## dropna values \ntrain_df.dropna(inplace = True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:42.977018Z","iopub.execute_input":"2025-04-12T10:13:42.977437Z","iopub.status.idle":"2025-04-12T10:13:43.031164Z","shell.execute_reply.started":"2025-04-12T10:13:42.977392Z","shell.execute_reply":"2025-04-12T10:13:43.030142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"##Train test splits\nfrom sklearn.model_selection import train_test_split\n\nx_cols  = train_df.columns.difference(target_cols)\ntrainx , testx , trainy , testy = train_test_split(train_df[x_cols] , train_df[target_cols ] , test_size = 0.25 , random_state = 0)\nprint(trainx.shape)\nprint(trainy.shape)\nprint(testx.shape)\nprint(testy.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:43.032301Z","iopub.execute_input":"2025-04-12T10:13:43.032700Z","iopub.status.idle":"2025-04-12T10:13:43.119787Z","shell.execute_reply.started":"2025-04-12T10:13:43.032662Z","shell.execute_reply":"2025-04-12T10:13:43.118803Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Model Building \nimport tensorflow as tf\nfrom tensorflow.keras.layers import Input, Embedding, Flatten, Concatenate, Dense, Normalization , Dropout\nfrom tensorflow.keras.models import Model\n\n\nembedding_dim = 16 \ncategorical_inputs = []\nembeddings = []\n\nfor col in categorical_cols:\n    vocab_size = vocab_sizes[col]\n    inp = Input(shape =(1,) , name = f'{col}_input')\n    categorical_inputs.append(inp)\n    emb = Embedding(input_dim = vocab_size , output_dim = embedding_dim)(inp)\n    flatten = Flatten()(emb)\n    embeddings.append(flatten)\n\nconcatenated_embeddings = Concatenate()(embeddings)\n\n## Numericals ###\nnumerical_inputs = Input(shape = (len(numerical_cols) ,) , name = 'numerical_input')\nnorm_layer = Normalization()\nnorm_layer.adapt(trainx[numerical_cols].values)\nnormalized_num = norm_layer(numerical_inputs)\n\n##concatenate ##\ncombined = Concatenate()([concatenated_embeddings , normalized_num])\n\n\nx = Dense(1024, activation='relu')(combined)\nx = Dropout(0.40)(x)\nx = Dense(512, activation='relu')(x)\nx = Dropout(0.40)(x)\nx = Dense(256, activation='relu')(x)\nx = Dropout(0.40)(x)\nx = Dense(64, activation='relu')(x)\noutput = Dense(3)(x)\n\n# Building Model ##\n\nmodel = Model(inputs = categorical_inputs + [numerical_inputs] , outputs = output)\n\nmodel.compile(loss = 'mse' , metrics = ['mae'] , optimizer = 'adam')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:43.120811Z","iopub.execute_input":"2025-04-12T10:13:43.121168Z","iopub.status.idle":"2025-04-12T10:13:46.480204Z","shell.execute_reply.started":"2025-04-12T10:13:43.121138Z","shell.execute_reply":"2025-04-12T10:13:46.478974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(model.summary())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:46.481288Z","iopub.execute_input":"2025-04-12T10:13:46.482070Z","iopub.status.idle":"2025-04-12T10:13:46.526223Z","shell.execute_reply.started":"2025-04-12T10:13:46.481998Z","shell.execute_reply":"2025-04-12T10:13:46.525161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.callbacks import EarlyStopping\n\nmodel_checkpoint =  ModelCheckpoint('best_model.keras' , save_best_only = True , mode = 'min' , monitor='val_loss' , verbose = 1)\nearly_stopping = EarlyStopping(patience=12, restore_best_weights=True, monitor='val_loss' , verbose = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:46.527232Z","iopub.execute_input":"2025-04-12T10:13:46.527600Z","iopub.status.idle":"2025-04-12T10:13:46.533564Z","shell.execute_reply.started":"2025-04-12T10:13:46.527572Z","shell.execute_reply":"2025-04-12T10:13:46.532718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## preparing dataset for model trainiong ##\nx_train_cat =  [trainx[col].values.reshape(-1 , 1) for col in categorical_cols]\nx_train_num = trainx[numerical_cols].values\nx_train = x_train_cat + [x_train_num]\ny_train = trainy.values\n\nx_val_cat = [testx[col].values.reshape(-1 , 1) for col in categorical_cols] \nx_val_num = testx[numerical_cols].values\nx_val = x_val_cat + [x_val_num]\ny_val = testy.values","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:46.534715Z","iopub.execute_input":"2025-04-12T10:13:46.535089Z","iopub.status.idle":"2025-04-12T10:13:46.558448Z","shell.execute_reply.started":"2025-04-12T10:13:46.535048Z","shell.execute_reply":"2025-04-12T10:13:46.557248Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(x_train , y_train , validation_data = (x_val , y_val) , epochs = 1000 , \n                    batch_size = 256, verbose = 1, callbacks = [model_checkpoint , early_stopping])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:13:46.561659Z","iopub.execute_input":"2025-04-12T10:13:46.562200Z","iopub.status.idle":"2025-04-12T10:20:45.284576Z","shell.execute_reply.started":"2025-04-12T10:13:46.562166Z","shell.execute_reply":"2025-04-12T10:20:45.283591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Preprocess test data (ensure categoricals are label-encoded)\n\nfor col in categorical_cols:\n    # Handle NaNs and unseen categories\n    test[col] = test[col].astype(str).fillna('missing')\n\n    le = label_encoders[col]\n    known_classes = set(le.classes_)\n\n    # Replace unseen labels with 'missing'\n    test[col] = test[col].apply(lambda x: x if x in known_classes else 'missing')\n\n    # Ensure 'missing' exists in label encoder\n    if 'missing' not in le.classes_:\n        le.classes_ = np.append(le.classes_, 'missing')\n\n    test[col] = le.transform(test[col])\n\n\n# Prepare test inputs\nX_test_cat = [test[col].values.reshape(-1, 1) for col in categorical_cols ]\nX_test_num = test[numerical_cols].values\nX_test = X_test_cat + [X_test_num]\n\n# Generate predictions\npredictions = model.predict(X_test)\nsubmission[['x_1', 'y_1', 'z_1']] = predictions\n# submission.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:20:45.285986Z","iopub.execute_input":"2025-04-12T10:20:45.286359Z","iopub.status.idle":"2025-04-12T10:20:45.930326Z","shell.execute_reply.started":"2025-04-12T10:20:45.286317Z","shell.execute_reply":"2025-04-12T10:20:45.928969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission[['ID', 'resname', 'resid', 'x_1', 'y_1', 'z_1']].head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:20:45.931712Z","iopub.execute_input":"2025-04-12T10:20:45.932120Z","iopub.status.idle":"2025-04-12T10:20:45.948840Z","shell.execute_reply.started":"2025-04-12T10:20:45.932082Z","shell.execute_reply":"2025-04-12T10:20:45.947344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create x_2-x_5, y_2-y_5, z_2-z_5 columns with same values as x_1/y_1/z_1\nfor i in range(2, 6):\n    submission[f'x_{i}'] = submission['x_1']\n    submission[f'y_{i}'] = submission['y_1']\n    submission[f'z_{i}'] = submission['z_1']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:20:45.950065Z","iopub.execute_input":"2025-04-12T10:20:45.950482Z","iopub.status.idle":"2025-04-12T10:20:45.976163Z","shell.execute_reply.started":"2025-04-12T10:20:45.950442Z","shell.execute_reply":"2025-04-12T10:20:45.974829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission['ID'] = submission['ID'].astype(str) + '_' + submission['resid'].astype(str)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:20:45.977242Z","iopub.execute_input":"2025-04-12T10:20:45.977861Z","iopub.status.idle":"2025-04-12T10:20:45.991742Z","shell.execute_reply.started":"2025-04-12T10:20:45.977815Z","shell.execute_reply":"2025-04-12T10:20:45.990264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.to_csv('submission.csv' , index = False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-12T10:20:45.993082Z","iopub.execute_input":"2025-04-12T10:20:45.993499Z","iopub.status.idle":"2025-04-12T10:20:46.062636Z","shell.execute_reply.started":"2025-04-12T10:20:45.993461Z","shell.execute_reply":"2025-04-12T10:20:46.061699Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}