{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom scipy.stats import rankdata\nimport os\nfrom copy import deepcopy as dc\n\nf = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')[['image_name']]\nn = f.shape[0]\ncols = {}\nfiles = [os.path.join('/kaggle/input/melanoma', file) for file in os.listdir('/kaggle/input/melanoma')] + [os.path.join('/kaggle/input/melanoma-public', file) for file in os.listdir('/kaggle/input/melanoma-public')]\nfor i, filename in enumerate(files):\n    cols[filename] = f'target_{i}'\n    ff = pd.read_csv(filename)\n    ff.columns = ['image_name', f'target_{i}']\n    ff[f'target_{i}'] = rankdata(ff[f'target_{i}'].values.tolist())\n    ff[f'target_{i}'] = ff[f'target_{i}'].apply(lambda x: (x-1)/(n-1))\n    f = f.merge(ff, on='image_name')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"f.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f['target'] = f[cols.values()].mean(axis=1)\nf[['image_name', 'target']].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f[['image_name', 'target']].to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Try to see which ones are the further from the other ones (probably more false)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def calculate_mse(y, df):\n    for col in cols.values():\n        df[col] = (df[col]-y)**2\n    return df.sum(axis=1)\ndic_errors = {}\nfor file, col in cols.items():\n    y = f[col]\n    error = calculate_mse(y,f[cols.values()])\n    dic_errors[file] = error","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"err_df = pd.DataFrame(dic_errors).transpose().sum(axis=1).sort_values(ascending=False)\nerr_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"N=4","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"biggest_error = err_df.index.tolist()[:N]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"biggest_error","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Seems we can exclude submissions 11 and 12...","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"f[f'target_wo_{N}'] = f[[c for k,c in cols.items() if k not in biggest_error]].mean(axis=1)\nf[['image_name', f'target_wo_{N}']].to_csv(f'sub_wo_{N}.csv', index=False, header=['image_name', 'target'])\nname_col = f'target_wo_{N}'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Let's do it intelligent : for each row take minimal distance","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"min_dist = pd.DataFrame(dic_errors).idxmin(axis=1)\nmin_vals = []\nfor i, sub in min_dist.iteritems():\n    min_vals.append(f.loc[i,cols[sub]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f['target_arg_min'] = min_vals","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f[['image_name', 'target_arg_min']].to_csv('sub_argmin.csv', index=False, header=['image_name', 'target'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Other option : for each row mean value of all but the N+1 furthest","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"vals = []\nfor i, row in pd.DataFrame(dic_errors).iterrows():\n    vals.append(f.loc[i,[cols[sub] for sub in row.sort_values(ascending=False).index.tolist()[N+1:]]].mean())\nf['target_mean_min'] = vals\nf[['image_name', 'target_mean_min']].to_csv('sub_mean_min.csv', index=False, header=['image_name', 'target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f['global_sub']= f[[name_col, 'target_mean_min']].mean(axis=1)\nf[['image_name', 'global_sub']].to_csv('global_sub.csv', index=False, header=['image_name', 'target'])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}