{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom PIL import Image\nfrom tqdm import tqdm\nfrom sklearn import preprocessing\nfrom sklearn.model_selection import StratifiedKFold,cross_val_score\nfrom sklearn.metrics import roc_auc_score\n\nimport lightgbm as lgb","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\nsample = pd.read_csv('../input/siim-isic-melanoma-classification/sample_submission.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['sex'] = train['sex'].fillna('na')\ntrain['age_approx'] = train['age_approx'].fillna(0)\ntrain['anatom_site_general_challenge'] = train['anatom_site_general_challenge'].fillna('na')\n\ntest['sex'] = test['sex'].fillna('na')\ntest['age_approx'] = test['age_approx'].fillna(0)\ntest['anatom_site_general_challenge'] = test['anatom_site_general_challenge'].fillna('na')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"trn_images = train['image_name'].values\ntrn_sizes = np.zeros((trn_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(trn_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/train/', f'{img_path}.jpg'))\n    trn_sizes[i] = np.array([img.size[0],img.size[1]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images = test['image_name'].values\ntest_sizes = np.zeros((test_images.shape[0],2))\nfor i, img_path in enumerate(tqdm(test_images)):\n    img = Image.open(os.path.join('../input/siim-isic-melanoma-classification/jpeg/test/', f'{img_path}.jpg'))\n    test_sizes[i] = np.array([img.size[0],img.size[1]])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train['w'] = trn_sizes[:,0]\ntrain['h'] = trn_sizes[:,1]\ntest['w'] = test_sizes[:,0]\ntest['h'] = test_sizes[:,1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"le = preprocessing.LabelEncoder()\n\ntrain.sex = le.fit_transform(train.sex)\ntrain.anatom_site_general_challenge = le.fit_transform(train.anatom_site_general_challenge)\ntest.sex = le.fit_transform(test.sex)\ntest.anatom_site_general_challenge = le.fit_transform(test.anatom_site_general_challenge)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = lgb.LGBMRegressor(n_estimators=500)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feature_names = ['sex','age_approx','anatom_site_general_challenge','w','h']\nycol = ['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test['target'] = 0\n\nkfold = StratifiedKFold(n_splits=10, shuffle=True, random_state=0)\n\nfor fold_id, (trn_idx, val_idx) in enumerate(kfold.split(train[feature_names], train[ycol])):\n    X_train = train.iloc[trn_idx][feature_names]\n    Y_train = train.iloc[trn_idx][ycol]\n\n    X_val = train.iloc[val_idx][feature_names]\n    Y_val = train.iloc[val_idx][ycol]\n\n    print('\\nFold_{} Training ================================\\n'.format(fold_id+1))\n\n    lgb_model = model.fit(X_train,\n                          Y_train,\n                          eval_names=['train', 'valid'],\n                          eval_set=[(X_train, Y_train), (X_val, Y_val)],\n                          verbose=100,\n                          eval_metric='auc',\n                          early_stopping_rounds=100)\n\n    pred_test = lgb_model.predict(test[feature_names], num_iteration=lgb_model.best_iteration_)\n    \n    test['target'] += pred_test / kfold.n_splits","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"lgb_model.feature_importances_","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample.target = test.target","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}