{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\nfrom tqdm import tqdm_notebook as bar\n\n# Any results you write to the current directory are saved as output.\ntrain_meta = pd.read_csv('../input/train.csv')\ntest_meta = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Mapping each siRNA and plate to 'groups'.\n\nThanks to [@zaharchikishev's kernel](https://www.kaggle.com/zaharch/keras-model-boosted-with-plates-leak/notebook), we've met another milestone for this competition.\n\nIn short, in this competition, siRNAs are treated on the plate in a group-wise manner, \n\nwhich means that we can split the whole problem predicting one from 1108 siRNAs into 4 subproblems, each of them predicting one from 1108 / 4 = 277 siRNAs!\n\nIn other words, there are 'four groups of siRNAs' (say Group 1, 2, 3 and 4), and within an experiment, each of the 'four groups of siRNAs' will have one-to-one matching to 'four plates of experiments'.\n\nIn theory, there will be 24 types of such one-to-one matchings, but it seems that only four of 24 possible one-to-one mappings were used in this competition (3 shown in training set, one more in test set, according to [@zaharchikishev's kernel](https://www.kaggle.com/zaharch/keras-model-boosted-with-plates-leak/notebook)).\n\nSo, in this kernel, I will simply focus on producing sirna-to-groups and plate-to-groups mapping table."},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_sirna_set(experiment, plate):\n    return set(train_meta[(train_meta.experiment == experiment)  & (train_meta.plate == plate)].sirna.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Create template siRNA groups, each of which contains 277 siRNAs.\ntemplate_sirna_groups = [get_sirna_set('HEPG2-01', plate) for plate in range(1, 5)]\n\nfor exp in train_meta.experiment.unique():\n    for plate in range(1, 5):\n        sirna_set = get_sirna_set(exp, plate)\n        \n        most_similar_group, max_similarity = None, -1\n        for i, template in enumerate(template_sirna_groups):\n            similarity = len(sirna_set & template)\n            if similarity > max_similarity:\n                most_similar_group, max_similarity = i, similarity\n            \n        if len(template_sirna_groups[most_similar_group]) != 277:\n            # Try to expand our template siRNA groups.\n            template_sirna_groups[most_similar_group] |= sirna_set\n            \nfor template in template_sirna_groups:\n    print(len(template))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_sirna_group(experiment, plate, template_sirna_groups=template_sirna_groups):\n    sirna_set = get_sirna_set(experiment, plate)\n    \n    for i, group in enumerate(template_sirna_groups, 1):\n        if sirna_set.issubset(group):\n            return i","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sirna_group = {\n    'sirna': [],\n    'group': [],\n}\n\nfor i, template in enumerate(template_sirna_groups, 1):\n    for sirna in template:\n        sirna_group['sirna'].append(sirna)\n        sirna_group['group'].append(i)\n\nsirna_group = pd.DataFrame(sirna_group).sort_values('sirna').reset_index()[['sirna', 'group']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sirna_group.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sirna_group.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sirna_group.to_csv('sirna_groups.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"experiment_plate_group = {\n    'experiment': [],\n    'plate': [],\n    'group': [],\n}\n\nfor experiment in train_meta.experiment.unique():\n    for plate in range(1, 5):\n        group = get_sirna_group(experiment, plate)\n        \n        experiment_plate_group['experiment'].append(experiment)\n        experiment_plate_group['plate'].append(plate)\n        experiment_plate_group['group'].append(group)\n\nexperiment_plate_group = pd.DataFrame(experiment_plate_group)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"experiment_plate_group.to_csv('plate_groups.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's see how many siRNA Group-to-plate mappings were used."},{"metadata":{"trusted":true},"cell_type":"code","source":"mapping_types = []\n\nfor experiment in train_meta.experiment.unique():\n    groups = experiment_plate_group[experiment_plate_group.experiment == experiment].group.values\n    mapping_types.append(''.join(map(str, groups)))\n    print(experiment, groups)\n    \nprint('\\nThere are %d unique sirna group-to-plate mapping types in total.' % len(set(mapping_types)))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Hope these tables help!"}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}