{"cells":[{"metadata":{},"cell_type":"markdown","source":"The purpose of this kernel is just to create a few datasets that can be used for further exploration and modeling in other kernels. ","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"It turns out that just the image metadata and and the image size contains a lot of useful information We'll start by creating feature-engineered datasets with just that information. The approach here follows the one in this notebook: https://www.kaggle.com/zzy990106/lgb-meta-data-image-size","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import StratifiedKFold,cross_val_score\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsample = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].fillna('na')\ntrain['age_approx'] = train['age_approx'].fillna(0)\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].fillna('na')\n\ntest['sex'] = test['sex'].fillna('na')\ntest['age_approx'] = test['age_approx'].fillna(0)\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].fillna('na')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trn_images = train['image_name'].values\ntrn_sizes = np.zeros((trn_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(trn_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    trn_sizes[i] = np.array([img.size[0],img.size[1]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images = test['image_name'].values\ntest_sizes = np.zeros((test_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(test_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    test_sizes[i] = np.array([img.size[0],img.size[1]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['w'] = trn_sizes[:,0]\ntrain['h'] = trn_sizes[:,1]\ntest['w'] = test_sizes[:,0]\ntest['h'] = test_sizes[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"le = preprocessing.LabelEncoder()\n\ntrain.sex = le.fit_transform(train.sex)\ntrain.anatom_site_general_challenge = le.fit_transform(train.anatom_site_general_challenge)\ntest.sex = le.fit_transform(test.sex)\ntest.anatom_site_general_challenge = le.fit_transform(test.anatom_site_general_challenge)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_names = ['sex','age_approx','anatom_site_general_challenge','w','h']\nycol = ['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[feature_names + ycol].to_csv('train_meta_size.csv', index=False)\ntest[feature_names ].to_csv('test_meta_size.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"The problem with the above approach is that we have very different distribution of missing values in train and test sets, so any algorithms that are sensitive to those discrepancies will lead to difference between the local CV and LB. We'll try to do somethign a bit more sophisticated now. ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"np.unique(train.diagnosis.values, return_counts=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"cols = ['sex', 'age_approx', 'anatom_site_general_challenge']\n\ntrain_test = train[cols].append(test[cols])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_test.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_test['age_approx'].mean()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_test['age_approx'] = train_test['age_approx'].fillna(train_test['age_approx'].mean())#float\ntrain_test['sex'] = train_test['sex'].fillna(train_test['sex'].value_counts().index[0])\ntrain_test['anatom_site_general_challenge'] = train_test['anatom_site_general_challenge'].fillna(train_test['anatom_site_general_challenge'].value_counts().index[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[cols] = train_test[:train.shape[0]][cols].values\ntest[cols] = train_test[train.shape[0]:][cols].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trn_images = train['image_name'].values\ntrn_sizes = np.zeros((trn_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(trn_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    trn_sizes[i] = np.array([img.size[0],img.size[1]])\n    \n    \ntest_images = test['image_name'].values\ntest_sizes = np.zeros((test_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(test_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    test_sizes[i] = np.array([img.size[0],img.size[1]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['w'] = trn_sizes[:,0]\ntrain['h'] = trn_sizes[:,1]\ntest['w'] = test_sizes[:,0]\ntest['h'] = test_sizes[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"le = preprocessing.LabelEncoder()\n\nle.fit(train_test.sex)\n\ntrain.sex = le.transform(train.sex)\ntest.sex = le.transform(test.sex)\n\nle = preprocessing.LabelEncoder()\n\nle.fit(train_test.anatom_site_general_challenge)\n\ntrain.anatom_site_general_challenge = le.transform(train.anatom_site_general_challenge)\ntest.anatom_site_general_challenge = le.transform(test.anatom_site_general_challenge)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train[feature_names + ycol].to_csv('train_meta_size_2.csv', index=False)\ntest[feature_names ].to_csv('test_meta_size_2.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ycol","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"We'll now add metafeatures from Chris Deotte's TF kernel:","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"oof_c = pd.read_csv('../input/triple-stratified-kfold-with-tfrecords/oof.csv')\nsubmission_c = pd.read_csv('../input/triple-stratified-kfold-with-tfrecords/submission.csv')\noof_c.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del oof_c['target']\noof_c.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"oof_c.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_2 = train[train['image_name'].isin(oof_c['image_name'].values)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_2 = train_2.merge(oof_c, on='image_name')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_names.append('pred')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission_c.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['pred'] = submission_c['target']\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ycol","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_2[feature_names + ['fold'] + ycol].to_csv('train_meta_size_3.csv', index=False)\ntest[feature_names ].to_csv('test_meta_size_3.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_2[feature_names + ['fold'] + ycol].head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test[feature_names]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_32 = np.load('../input/siimisic-melanoma-resized-images/x_train_32.npy')/255\ntest_32 = np.load('../input/siimisic-melanoma-resized-images/x_test_32.npy')/255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_32 = train_32.reshape((train_32.shape[0], 32*32*3))\ntest_32 = test_32.reshape((test_32.shape[0], 32*32*3))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"columns = [f'c_{i}' for i in range(3072)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_32 = pd.DataFrame(data = train_32, columns=columns)\ntest_32 = pd.DataFrame(data = test_32, columns=columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_32['target'] = train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_32.to_csv('train_32.csv', index=False)\ntest_32.to_csv('test_32.csv', index=False)\nnp.save('columns_32', columns)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}