{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59094,"databundleVersionId":7010844,"sourceType":"competition"},{"sourceId":6705012,"sourceType":"datasetVersion","datasetId":3864332},{"sourceId":6898544,"sourceType":"datasetVersion","datasetId":3962716},{"sourceId":7073625,"sourceType":"datasetVersion","datasetId":3981209},{"sourceId":7712331,"sourceType":"datasetVersion","datasetId":4441094}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Neural network model \nThis notebook covers the training of neural network used for 4th place solution.  ","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5"}},{"cell_type":"code","source":"import time\nimport pandas as pd\nimport tensorflow as tf\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:04:27.085155Z","iopub.execute_input":"2023-12-05T07:04:27.085432Z","iopub.status.idle":"2023-12-05T07:04:39.23934Z","shell.execute_reply.started":"2023-12-05T07:04:27.085407Z","shell.execute_reply":"2023-12-05T07:04:39.238057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read the data","metadata":{}},{"cell_type":"code","source":"data = pd.read_parquet(\"/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet\")\nid_map = pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/id_map.csv\")\nsample_submission=  pd.read_csv(\"/kaggle/input/open-problems-single-cell-perturbations/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:05:10.017613Z","iopub.execute_input":"2023-12-05T07:05:10.017981Z","iopub.status.idle":"2023-12-05T07:05:13.562823Z","shell.execute_reply.started":"2023-12-05T07:05:10.017951Z","shell.execute_reply":"2023-12-05T07:05:13.562011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:05:13.564262Z","iopub.execute_input":"2023-12-05T07:05:13.564566Z","iopub.status.idle":"2023-12-05T07:05:13.610046Z","shell.execute_reply.started":"2023-12-05T07:05:13.56454Z","shell.execute_reply":"2023-12-05T07:05:13.609176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocess ","metadata":{}},{"cell_type":"code","source":"def add_columns(data, id_map):\n    id_map['SMILES'] = id_map['sm_name'].map(data.set_index('sm_name')['SMILES'].to_dict())\n    id_map['sm_lincs_id'] = id_map['sm_name'].map( data.set_index('sm_name')[\"sm_lincs_id\"].to_dict())\n    return id_map","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:05:16.902308Z","iopub.execute_input":"2023-12-05T07:05:16.903193Z","iopub.status.idle":"2023-12-05T07:05:16.908674Z","shell.execute_reply.started":"2023-12-05T07:05:16.90316Z","shell.execute_reply":"2023-12-05T07:05:16.907548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"add_columns(data, id_map)","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:05:18.331727Z","iopub.execute_input":"2023-12-05T07:05:18.332106Z","iopub.status.idle":"2023-12-05T07:05:18.419749Z","shell.execute_reply.started":"2023-12-05T07:05:18.332073Z","shell.execute_reply":"2023-12-05T07:05:18.418776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\n# Suppress FutureWarnings related to is_sparse\nwarnings.simplefilter(action='ignore', category=FutureWarning)\n\n\nfeatures_columns_1 = [\"cell_type\", \"sm_name\"]\nfeatures_columns_2 = [\"cell_type\", \"sm_lincs_id\"]\nfeatures_columns_3 = [\"cell_type\", \"SMILES\"]\nlabels_columns=[\"cell_type\",\"sm_name\",\"sm_lincs_id\",\"SMILES\",\"control\"]\nlabels = data.drop(columns=labels_columns)\n\nfeatures_1 = pd.DataFrame(data, columns=features_columns_1)\nfeatures_2 = pd.DataFrame(data, columns=features_columns_2)\nfeatures_3 = pd.DataFrame(data, columns=features_columns_3)\n\ntest_1 = pd.DataFrame(id_map, columns=features_columns_1)\ntest_2 = pd.DataFrame(id_map, columns=features_columns_2)\ntest_3 = pd.DataFrame(id_map, columns=features_columns_3)\n\n# Onhot_encode features\nencoder_1 = OneHotEncoder()\nencoder_2 = OneHotEncoder()\nencoder_3 = OneHotEncoder()\n\n# Full training and test data\nfull_feature_1 = pd.DataFrame(encoder_1.fit_transform(features_1).toarray(), columns=encoder_1.get_feature_names_out(features_columns_1))\nfull_feature_2 = pd.DataFrame(encoder_2.fit_transform(features_2).toarray(), columns=encoder_2.get_feature_names_out(features_columns_2))\nfull_feature_3 = pd.DataFrame( encoder_3.fit_transform(features_3).toarray(), columns=encoder_3.get_feature_names_out(features_columns_3))\n\nfull_test_1 = encoder_1.transform(test_1).toarray()\nfull_test_2 = encoder_2.transform(test_2).toarray()\nfull_test_3 = encoder_3.transform(test_3).toarray()\n\n# Normalized the labels\nscaler_label = StandardScaler()\nlabels_df = pd.DataFrame(scaler_label.fit_transform(labels), columns=labels.columns)","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:07:56.204398Z","iopub.execute_input":"2023-12-05T07:07:56.204796Z","iopub.status.idle":"2023-12-05T07:07:58.716034Z","shell.execute_reply.started":"2023-12-05T07:07:56.204766Z","shell.execute_reply":"2023-12-05T07:07:58.715032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"physical_devices = tf.config.list_physical_devices('GPU')\nif physical_devices:\n    tf.config.set_visible_devices(physical_devices, 'GPU')\n    print(\"Using GPU:\", physical_devices)\nelse:\n    print(\"No GPU available. Switching to CPU.\")\nstrategy = tf.distribute.MirroredStrategy()","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:07:58.788109Z","iopub.execute_input":"2023-12-05T07:07:58.788808Z","iopub.status.idle":"2023-12-05T07:07:58.796307Z","shell.execute_reply.started":"2023-12-05T07:07:58.788779Z","shell.execute_reply":"2023-12-05T07:07:58.795379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create model ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers, Model\n\ninput_shape=(152,)\nbatch_size=32\nwith strategy.scope():\n    def create_multimodal_model(input_shape):\n        \n        # Input layers for the three parts\n        input_sm_name = layers.Input(shape=input_shape, batch_size=batch_size ,name=\"cell_type-sm_name\")\n        input_sm_lincs = layers.Input(shape=input_shape, batch_size=batch_size, name=\"cell_type-sm_lincs\")\n        input_smiles = layers.Input(shape=input_shape, batch_size=batch_size, name=\"cell_type-SMILES\")\n\n        # Function to create  sub-models\n        def create_submodel(input_layer):\n            x = layers.Dense(512, activation=\"relu\")(input_layer)\n            x = layers.Dense(512, activation=\"relu\")(x)\n            x = layers.Dense(256, activation=\"relu\")(x)\n            output = layers.Dense(128, activation=\"relu\")(x)\n            return output\n\n        # Call the function to create sub-models for each input\n        sm_name_output = create_submodel(input_sm_name)\n        sm_lincs_output = create_submodel(input_sm_lincs)\n        smiles_output = create_submodel(input_smiles)\n\n        # Concatenate the outputs\n        concat = layers.Concatenate(name=\"concatenate_layer\")([sm_name_output, sm_lincs_output, smiles_output])\n\n        # Construct output layers\n        combined_dropout = layers.Dropout(0.5)(concat)\n        x = layers.Dense(256, activation=\"relu\")(combined_dropout)\n        output_layer = layers.Dense(18211, \"linear\", name='output_layer')(x)\n\n        # Create the final model\n        model = Model(inputs=[input_sm_name, input_sm_lincs, input_smiles], outputs=output_layer)\n\n        return model","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:08:02.865281Z","iopub.execute_input":"2023-12-05T07:08:02.866142Z","iopub.status.idle":"2023-12-05T07:08:02.970718Z","shell.execute_reply.started":"2023-12-05T07:08:02.866109Z","shell.execute_reply":"2023-12-05T07:08:02.969889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train model ","metadata":{}},{"cell_type":"code","source":"initial_time = time.time()\nmrrrmse_max = 0.5999\nepochs = 1  \nmodel_predictions = []\nscore_mrrmse = []\n\nfor i in range(50):\n    start_time = time.time()\n    total_epochs = 0\n    random_seed = np.random.randint(1, 10000)\n    shuffled_indices = np.random.permutation(len(full_feature_1))\n    shuffled_X_1 = full_feature_1.values[shuffled_indices]\n    shuffled_X_2 = full_feature_2.values[shuffled_indices]\n    shuffled_X_3 = full_feature_3.values[shuffled_indices]\n    shuffled_y = labels_df.values[shuffled_indices]\n\n    with strategy.scope():\n        full_dataset = tf.data.Dataset.from_tensor_slices(\n            ({\"cell_type-sm_name\": shuffled_X_1, \"cell_type-sm_lincs\": shuffled_X_2, \"cell_type-SMILES\": shuffled_X_3},\n             shuffled_y)).batch(batch_size).prefetch(buffer_size=tf.data.AUTOTUNE)\n\n        model_1 = create_multimodal_model(input_shape)\n        model_1.compile(loss='mae', metrics=['mae'], optimizer='adam')\n        model_1.fit(full_dataset, epochs=230, verbose=0)\n        total_epochs += 230\n        print('Trained 230 epochs')\n\n    while True:\n        with strategy.scope():\n            model_1.fit(full_dataset, epochs=epochs, verbose=0)\n        total_epochs += epochs\n\n        y_pred = model_1.predict([shuffled_X_1, shuffled_X_2, shuffled_X_3], batch_size=1, verbose=0)\n        mrrmse =np.mean(np.sqrt(np.mean(np.square(scaler_label.inverse_transform(shuffled_y) - scaler_label.inverse_transform(y_pred)), axis=1)))\n           \n        if mrrmse <= mrrrmse_max:\n            # Predict on test data\n            pred = model_1.predict([full_test_1, full_test_2, full_test_3], batch_size=1)\n            prediction = scaler_label.inverse_transform(pred)\n            model_predictions.append(prediction)\n            score_mrrmse.append(mrrmse)\n            print(f'loop_{i} finished with mrrmse {mrrmse}, total epochs {total_epochs}')     \n            print(f\"\\nTotal Execution Time for loop{i}: {time.time() - start_time} seconds\")\n            break\n    \n        if total_epochs > 300:\n            print('epochs above 300 epochs: ', total_epochs)\n            break\n\nprint(f\"\\nTotal Execution Time: {time.time() - initial_time} seconds\")","metadata":{"execution":{"iopub.status.busy":"2023-12-05T07:08:07.340162Z","iopub.execute_input":"2023-12-05T07:08:07.340826Z","iopub.status.idle":"2023-12-05T09:18:54.449804Z","shell.execute_reply.started":"2023-12-05T07:08:07.340791Z","shell.execute_reply":"2023-12-05T09:18:54.448888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction","metadata":{}},{"cell_type":"code","source":"preds = np.mean(model_predictions, axis=0 )\npred_df= pd.DataFrame(preds , columns=sample_submission.columns[1:])\npred_df.insert(0, 'id' , range(0,255))\npred_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-05T09:26:43.549132Z","iopub.execute_input":"2023-12-05T09:26:43.54957Z","iopub.status.idle":"2023-12-05T09:26:51.620284Z","shell.execute_reply.started":"2023-12-05T09:26:43.549535Z","shell.execute_reply":"2023-12-05T09:26:51.619234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot predictions","metadata":{}},{"cell_type":"markdown","source":"Thanks to @awater1223 for sharing the work https://www.kaggle.com/code/awater1223/op2-00-basic-metadata-eda ","metadata":{}},{"cell_type":"code","source":"control_ids = [\n#     'LSM-36361'  # DMSO\n    'LSM-43181',  # Belinostat\n    'LSM-6303',  # Dabrafenib\n]\n\nprivte_ids = [\n    'LSM-45710', 'LSM-4062', \n    'LSM-2193',  #  'forskolin' -> 'Colforsin'\n    'LSM-4105', 'LSM-4031', 'LSM-1099', 'LSM-45153', 'LSM-3822', 'LSM-4933', \n    'LSM-45630',  # 'KD-025' -> 'SLx-2119'\n    'LSM-6258', 'LSM-1023', 'LSM-2655', 'LSM-47602', 'LSM-3349', 'LSM-1020', 'LSM-1143',\n    'LSM-3828', 'LSM-1051', 'LSM-1120', 'LSM-5467', 'LSM-2292', 'LSM-43293', 'LSM-45437',\n    'LSM-2703', 'LSM-45831', 'LSM-1179', 'LSM-1199', 'LSM-1190', 'LSM-36374', 'LSM-5215',\n    'LSM-1195', 'LSM-45468', 'LSM-45410', 'LSM-47459', 'LSM-45663', 'LSM-45518', 'LSM-1062',\n    'LSM-3667',  # 'BRD-K74305673' -> 'IMD-0354',\n    'LSM-1032', 'LSM-5855', 'LSM-45988',\n    'LSM-24954',  # 'BRD-K98039984' -> 'Prednisolone'\n    'LSM-6286', 'LSM-45984', 'LSM-1124', 'LSM-1165', 'LSM-42802', 'LSM-1121', 'LSM-6308',\n    'LSM-1136', 'LSM-1186', 'LSM-45915', 'LSM-2621', 'LSM-5341', 'LSM-45724', 'LSM-2219',\n    'LSM-2936', 'LSM-3171', 'LSM-46889', 'LSM-2379', 'LSM-47132', 'LSM-47120', 'LSM-47437',\n    'LSM-1139', 'LSM-1144', 'LSM-4353', 'LSM-1210', 'LSM-5887', 'LSM-1025', 'LSM-5771', 'LSM-1132',\n    'LSM-1263',  # 'BRD-A04553218' -> 'Chlorpheniramine'\n    'LSM-1167',\n    'LSM-1194',  # 'BRD-A92800748' -> 'TIE2 Kinase Inhibitor'\n    'LSM-45948', 'LSM-45514', 'LSM-5430', 'LSM-2309', \n]\n\nlen(privte_ids)\n\npublic_ids = [\n    'LSM-43216', 'LSM-1050', 'LSM-45849', 'LSM-42800', 'LSM-1131', 'LSM-6335', 'LSM-1211',\n    'LSM-45239', 'LSM-1130', 'LSM-45786', 'LSM-5199', 'LSM-45281',\n    'LSM-6324', # 'ACY-1215' -> 'Ricolinostat'\n    'LSM-3309', 'LSM-1056', 'LSM-45591', 'LSM-46203', 'LSM-5662',\n    'LSM-47134',  # 'SB-2342' -> '5-(9-Isopropyl-8-methyl-2-morpholino-9H-purin-6-yl)pyrimidin-2-amine\t'\n    'LSM-45637', 'LSM-1127', 'LSM-46971', 'LSM-1172', 'LSM-46042', 'LSM-1101', 'LSM-45758',\n    'LSM-5218', 'LSM-2287', 'LSM-1014',\n    'LSM-1040', #  'fostamatinib' -> 'Tamatinib'\n    'LSM-1476;LSM-5290',\n    'LSM-45680',  # 'basimglurant' -> 'RG7090'\n    'LSM-4349',  # '5-iodotubercidin' -> 'IN1451'\n    'LSM-3425', 'LSM-45806',\n    'LSM-45616',  # 'SB-683698' -> 'TR-14035'\n    'LSM-1055',\n    'LSM-43281',  # 'C-646' -> 'STK219801'\n    'LSM-5690', 'LSM-1155', 'LSM-2499',\n    'LSM-2382',  # 'JTC-801' -> 'UNII-BXU45ZH6LI'\n    'LSM-45220', 'LSM-1037', 'LSM-1005', 'LSM-1180', 'LSM-36812',\n    'LSM-45924',  # 'filgotinib' -> 'GLPG0634'\n    'LSM-2013',  # 'TL-HRAS-61' -> TL_HRAS26'\n    'LSM-4738',\n]\n\ntrain_ids = [\n    'LSM-1027', 'LSM-1071', 'LSM-45916',\n    'LSM-4944',  # 'ixazomib' -> 'MLN 2238'\n    'LSM-47425',  # 'IWP-L6' -> 'Porcn Inhibitor III'\n    'LSM-1115',\n    'LSM-6237',  # 'CD-437' -> 'O-Demethylated Adapalene'\n    'LSM-1205', 'LSM-45574',\n    'LSM-4255',  # 'NVP-BEZ235' -> 'Dactolisib'\n    'LSM-1181', 'LSM-1158', 'LSM-2334', 'LSM-45496', 'LSM-1011',\n]\n\ndef classify_id(row):\n    if row['sm_lincs_id'] in privte_ids:\n        return 'Private'\n    elif row['sm_lincs_id'] in public_ids:\n        return 'Public'\n    elif row['sm_lincs_id'] in train_ids:\n        return 'Train'\n    elif row['sm_lincs_id'] in control_ids:\n        return 'Control'\n    else:\n        return 'Other'\n\nlb_split = data.apply(classify_id, axis=1)\ndata.insert(2, 'lb_split', lb_split)","metadata":{"execution":{"iopub.status.busy":"2023-12-04T12:39:36.614355Z","iopub.execute_input":"2023-12-04T12:39:36.614754Z","iopub.status.idle":"2023-12-04T12:39:37.600318Z","shell.execute_reply.started":"2023-12-04T12:39:36.614722Z","shell.execute_reply":"2023-12-04T12:39:37.599432Z"},"_kg_hide-output":false,"_kg_hide-input":false,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting `np.abs(prediction).mean` in scatter plot","metadata":{}},{"cell_type":"code","source":"df = pred_df.copy()\ndf = df.drop('id', axis=1)\ndf.insert(0, 'sm_name', id_map['sm_name'])  \ndf.insert(1, 'lb_split', id_map['sm_name'].map(data.set_index('sm_name')['lb_split'].to_dict()))\ndf = df.sort_values('sm_name')\nper_compound_mean_abs_gene = df.iloc[:, 2:].apply(lambda row: np.abs(row).mean(), axis=1)\ndf.insert(0, 'per_compound_mean_abs_gene', per_compound_mean_abs_gene)\n\npublic_test = df[df['lb_split'] == 'Public']\nprivate_test = df[df['lb_split'] == 'Private']\n\nplt.figure(figsize=(10, 6))\nplt.scatter(range(len(public_test)), public_test['per_compound_mean_abs_gene'], label='Public test compound ', color='blue', alpha=0.7)\nplt.scatter(range(len(private_test)), private_test['per_compound_mean_abs_gene'], label='Private_compound ', color='red', alpha=0.7)\nplt.xlabel('Compounds', fontsize=12)\nplt.ylabel('np.abs(prediction).mean', fontsize=12)\nplt.title('NN Public and Private Test compounds prediction', fontsize=15)\nplt.legend()\nplt.grid(True)\nplt.tight_layout()\n# plt.savefig('nn_lgbm.png')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-04T12:45:56.080661Z","iopub.execute_input":"2023-12-04T12:45:56.081059Z","iopub.status.idle":"2023-12-04T12:45:56.620145Z","shell.execute_reply.started":"2023-12-04T12:45:56.08103Z","shell.execute_reply":"2023-12-04T12:45:56.619117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Bar plot of average per compound `np.abs().mean` for de_train and predictions","metadata":{}},{"cell_type":"code","source":"df = data.copy()\ndf = df.sort_values(by='sm_name')\n\nper_compound_mean_abs_gene = df.iloc[:, 6:].apply(lambda row: np.abs(row).mean(), axis=1)\ndf.insert(0, 'per_compound_mean_abs_gene', per_compound_mean_abs_gene)\n\npublic_test_de = df[df['lb_split'] == 'Public']\nprivate_test_de = df[df['lb_split'] == 'Private']\n# control_de = df[df['lb_split'] == 'Control']\n# train_de = df[df['lb_split'] == 'Train']\n\n\nsm_name_public =  public_test['sm_name'].unique()\nsm_name_private =  private_test['sm_name'].unique()\nsm_name_private.sort\nsm_name_public.sort\n\naverage_per_compound_public_nn = public_test.groupby('sm_name')['per_compound_mean_abs_gene'].mean().reset_index()\naverage_per_compound_public_de = public_test_de.groupby('sm_name')['per_compound_mean_abs_gene'].mean().reset_index()\n\naverage_per_compound_private_nn = private_test.groupby('sm_name')['per_compound_mean_abs_gene'].mean().reset_index()\naverage_per_compound_private_de = private_test_de.groupby('sm_name')['per_compound_mean_abs_gene'].mean().reset_index()\n\n# Bar plot for 'Public' and 'private' test data with averaged per_compound_mean_abs_gene\nplt.figure(figsize=(40, 35))\n\n# Subplot 1\nplt.subplot(2, 1, 1)  # 1 row, 2 columns, subplot 1\nplt.bar(sm_name_public, average_per_compound_public_nn['per_compound_mean_abs_gene'], label='Averaged Public NN pred compounds', color='blue', alpha=0.5)\nplt.bar(sm_name_public, average_per_compound_public_de['per_compound_mean_abs_gene'], label='Averaged Public de_train compounds', color='green', alpha=0.5)\nplt.xlabel('Public Compounds', fontsize=20)\nplt.ylabel('Averaged Per Compound: abs().mean', fontsize=20)\nplt.title('Averaged abs().mean for Public Compounds', fontsize=30)\nplt.legend(fontsize='large')\nplt.grid(True)\nplt.xticks(rotation=90)\n\n# Subplot 2\nplt.subplot(2, 1, 2)  # 1 row, 2 columns, subplot 2\nplt.bar(sm_name_private, average_per_compound_private_nn['per_compound_mean_abs_gene'], label='Averaged Private NN pred compounds', color='blue', alpha=0.5)\nplt.bar(sm_name_private, average_per_compound_private_de['per_compound_mean_abs_gene'], label='Averaged Private de_train compounds', color='green', alpha=0.5)\nplt.xlabel('Private Test Compounds', fontsize=20)\nplt.ylabel('Averaged Per Compound: abs().mean', fontsize=20)\nplt.title('Averaged abs().mean for Private Compounds', fontsize=30)\nplt.legend(fontsize='large')\nplt.grid(True)\nplt.xticks(rotation=90)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-12-04T13:16:55.66508Z","iopub.execute_input":"2023-12-04T13:16:55.665748Z","iopub.status.idle":"2023-12-04T13:16:58.660127Z","shell.execute_reply.started":"2023-12-04T13:16:55.665717Z","shell.execute_reply":"2023-12-04T13:16:58.659125Z"},"trusted":true},"execution_count":null,"outputs":[]}]}