{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport math","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/human-protein-atlas-image-classification/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_target_list(str_target):\n    return [int(i) for i in str_target.split()]\n\ndef label_in_target_list(label, target_list):\n    if label in target_list:\n        return 1\n    return 0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"LABELS = list(range(28))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def train_test_split(df, size=0.1, random_seed=8430):\n    \"\"\"\n    df: pd.DataFame with columns ['Id', 'Target']\n    size: fraction for test set\n    random_seed:\n    \n    return: train and test datafame \n    \"\"\"\n    \n    df['Target_List'] = df['Target'].apply(lambda x: get_target_list(x))\n\n    for label in LABELS:\n        df[label] = df['Target_List'].apply(lambda tl: label_in_target_list(label, tl))\n    \n    sorted_counts_by_label = df[LABELS].sum().sort_values()\n\n    val_size = (sorted_counts_by_label * size).apply(lambda x: math.ceil(x))\n    \n    df['Test_Set'] = 0\n\n    counter = pd.Series(index=LABELS, data=0)\n    for label, total in val_size.items():\n        num_to_sample = total - counter.loc[label]\n        idx = df[(df['Test_Set'] == 0) & (df[label] == 1)].sample(\n            num_to_sample, random_state=random_seed).index\n        counter += df.loc[idx][LABELS].sum()\n        df.at[idx, 'Test_Set'] = 1\n\n    return df[df['Test_Set']==0][['Id', 'Target']], df[df['Test_Set']==1][['Id', 'Target']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, test_df = train_test_split(df, size=0.1, random_seed=3561)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.to_csv('train_df.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.to_csv('test_df.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}