{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install -q /kaggle/input/iterative-stratification/iterative-stratification-master/","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom fastai.vision.all import *\nimport pickle\nimport os\nimport warnings\n\nwarnings.filterwarnings('ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/knn-dataset/combined.csv')\ndf = df.sample(frac=1, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"labels = [str(i) for i in range(19)]\nfor x in labels: df[x] = df['Label'].apply(lambda r: int(x in r.split('|')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i,row in df.iterrows():\n    if row['cluster'] == -1:\n        df['cluster'].loc[i] = i + 7933","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_copy = df['cluster'].copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df), len(df_copy)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_copy = df_copy.drop_duplicates()\nlen(df_copy)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_copy = df.loc[df_copy.index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_counts(dfs):\n    unique_counts = {}\n    for lbl in labels:\n        unique_counts[lbl] = len(dfs[dfs.Label == lbl])\n\n    full_counts = {}\n    for lbl in labels:\n        count = 0\n        for row_label in dfs['Label']:\n            if lbl in row_label.split('|'): count += 1\n        full_counts[lbl] = count\n\n    counts = list(zip(full_counts.keys(), full_counts.values(), unique_counts.values()))\n    counts = np.array(sorted(counts, key=lambda x:-x[1]))\n    counts = pd.DataFrame(counts, columns=['label', 'full_count', 'unique_count'])\n    counts = counts.set_index('label').T\n    return counts","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"get_counts(df_copy)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs = df_copy.reset_index(drop=True)\nnfold = 5\nseed = 42\n\ny = dfs[labels].values\nX = dfs['ID'].values\n\ndfs['fold'] = np.nan\n\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\nmskf = MultilabelStratifiedKFold(n_splits=nfold, random_state=seed)\nfor i, (_, test_index) in enumerate(mskf.split(X, y)):\n    dfs.iloc[test_index, -1] = i\n    \ndfs['fold'] = dfs['fold'].astype('int')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_folds = dfs[['cluster', 'fold']].drop_duplicates()\nlen(df_folds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = pd.merge(df, df_folds, how='left', on='cluster')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.fold.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs1 = df[df['fold'] == 0]\nc1 = dfs1.cluster.unique().tolist()\nget_counts(dfs1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs1 = df[df['fold'] == 1]\nc2 = dfs1.cluster.unique().tolist()\nget_counts(dfs1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs1 = df[df['fold'] == 2]\nc3 = dfs1.cluster.unique().tolist()\nget_counts(dfs1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs1 = df[df['fold'] == 3]\nc4 = dfs1.cluster.unique().tolist()\nget_counts(dfs1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs1 = df[df['fold'] == 4]\nc5 = dfs1.cluster.unique().tolist()\nget_counts(dfs1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"assert len(set(c1 + c2 + c3 + c4 + c5)) == len(c1 + c2 + c3 + c4 + c5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df = df[['ID', 'Label', 'dataset', 'fold']]\ndf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.to_csv('hpa_folds_v1.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}