{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import das bibliotecas\nimport numpy as np\nimport pandas as pd\nimport cv2\nfrom glob import glob \nimport os\nimport matplotlib.pyplot as plt\nimport tqdm\n\nfrom sklearn.metrics import classification_report, log_loss, accuracy_score, roc_curve, auc, roc_auc_score\nfrom sklearn.model_selection import train_test_split\n\nimport xgboost as xgb\n\nnp.random.seed(42)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:46.382340Z","iopub.execute_input":"2021-08-19T10:52:46.382672Z","iopub.status.idle":"2021-08-19T10:52:47.313267Z","shell.execute_reply.started":"2021-08-19T10:52:46.382644Z","shell.execute_reply":"2021-08-19T10:52:47.312407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Realizando o a leitura do dataset\n\nLeitura do dataset, no momento estamos realizando testes com apenas 10 mil imagens.","metadata":{}},{"cell_type":"code","source":"path = \"/kaggle/input/histopathologic-cancer-detection/\" \nlabels = pd.read_csv(path + 'train_labels.csv')\ntrain_path = path + 'train/'","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:47.314720Z","iopub.execute_input":"2021-08-19T10:52:47.315073Z","iopub.status.idle":"2021-08-19T10:52:47.877109Z","shell.execute_reply.started":"2021-08-19T10:52:47.315037Z","shell.execute_reply":"2021-08-19T10:52:47.876075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'path': glob(os.path.join(train_path,'*.tif'))})\ndf['id'] = df.path.map(lambda x: ((x.split(\"n\")[-1].split('.')[0])[1:]))\ndf = df.merge(labels, on = \"id\")\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:47.881976Z","iopub.execute_input":"2021-08-19T10:52:47.882316Z","iopub.status.idle":"2021-08-19T10:52:50.714933Z","shell.execute_reply.started":"2021-08-19T10:52:47.882283Z","shell.execute_reply":"2021-08-19T10:52:50.714163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 96\nBATCH_SIZE = 128","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:50.716580Z","iopub.execute_input":"2021-08-19T10:52:50.716951Z","iopub.status.idle":"2021-08-19T10:52:50.721098Z","shell.execute_reply.started":"2021-08-19T10:52:50.716913Z","shell.execute_reply":"2021-08-19T10:52:50.720095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Distribuição das classes\");\n\nplt.pie(df['label'].value_counts(), labels=['Sem cancer',\n          'Com Cancer'], startangle=180, autopct='%1.1f', \n           colors=['#00ff99','#FF96A7'], shadow=True);\nplt.figure(figsize=(16,16));\nplt.show();","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:50.722596Z","iopub.execute_input":"2021-08-19T10:52:50.722967Z","iopub.status.idle":"2021-08-19T10:52:50.857912Z","shell.execute_reply.started":"2021-08-19T10:52:50.722931Z","shell.execute_reply":"2021-08-19T10:52:50.856834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:50.859402Z","iopub.execute_input":"2021-08-19T10:52:50.859748Z","iopub.status.idle":"2021-08-19T10:52:50.869293Z","shell.execute_reply.started":"2021-08-19T10:52:50.859713Z","shell.execute_reply":"2021-08-19T10:52:50.868471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:52:50.870622Z","iopub.execute_input":"2021-08-19T10:52:50.871201Z","iopub.status.idle":"2021-08-19T10:52:50.879396Z","shell.execute_reply.started":"2021-08-19T10:52:50.871153Z","shell.execute_reply":"2021-08-19T10:52:50.878497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = []\ny = []\nfor idx in tqdm.tqdm(range(df.shape[0])):\n\n    X.append(cv2.imread(df.iloc[idx]['path']))\n    y.append(df.iloc[idx]['label'])\n    if idx == 10000:\n        break\nX = np.array(X)\ny = np.array(y)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:08:06.668142Z","iopub.execute_input":"2021-08-19T11:08:06.668469Z","iopub.status.idle":"2021-08-19T11:08:24.620026Z","shell.execute_reply.started":"2021-08-19T11:08:06.668441Z","shell.execute_reply":"2021-08-19T11:08:24.619115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Primeira imagem do dataset\")\nimagem = cv2.imread(df['path'][0], cv2.IMREAD_GRAYSCALE)\nplt.imshow(imagem);","metadata":{"execution":{"iopub.status.busy":"2021-08-19T10:54:24.905742Z","iopub.execute_input":"2021-08-19T10:54:24.906013Z","iopub.status.idle":"2021-08-19T10:54:25.070585Z","shell.execute_reply.started":"2021-08-19T10:54:24.905987Z","shell.execute_reply":"2021-08-19T10:54:25.069742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dividindo o dataset\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:10:10.047374Z","iopub.execute_input":"2021-08-19T11:10:10.047722Z","iopub.status.idle":"2021-08-19T11:10:10.125756Z","shell.execute_reply.started":"2021-08-19T11:10:10.047692Z","shell.execute_reply":"2021-08-19T11:10:10.124803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sobre o Classificador\n\nO XGBoost  (e**X**treme **G**radient **Boost**ing) é um algoritmo de classificação (possui versão pra regressão também) baseado em árvore de decisão com Gradient Boosting (Aumento de gradiente).\n\n### Sobre Gradient Boosting\n\nGradient Boosting é uma técnica que utiliza de *ensemble* de modelos considerados fracos, no caso do XGBoost é utilizado árvore de decisão","metadata":{}},{"cell_type":"code","source":" xg_clf = xgb.XGBClassifier(objective ='reg:squarederror', colsample_bytree = 0.3, learning_rate = 0.1,\n                max_depth = 5, alpha = 10, n_estimators = 10)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:53:41.942441Z","iopub.execute_input":"2021-08-19T11:53:41.942782Z","iopub.status.idle":"2021-08-19T11:53:41.946640Z","shell.execute_reply.started":"2021-08-19T11:53:41.942753Z","shell.execute_reply":"2021-08-19T11:53:41.945817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = X_train.reshape(len(X_train),3 * 96 * 96)\nX_test = X_test.reshape(len(X_test),3 * 96 * 96)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:10:13.162708Z","iopub.execute_input":"2021-08-19T11:10:13.163041Z","iopub.status.idle":"2021-08-19T11:10:13.166825Z","shell.execute_reply.started":"2021-08-19T11:10:13.163010Z","shell.execute_reply":"2021-08-19T11:10:13.165999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xg_clf.fit(X_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:53:44.098190Z","iopub.execute_input":"2021-08-19T11:53:44.098532Z","iopub.status.idle":"2021-08-19T11:54:26.970092Z","shell.execute_reply.started":"2021-08-19T11:53:44.098501Z","shell.execute_reply":"2021-08-19T11:54:26.969287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualizado as métricas","metadata":{}},{"cell_type":"code","source":"predictions  = xg_clf.predict(X_test)\nfalse_positive_rate, true_positive_rate, threshold = roc_curve(y_test, predictions)\narea_under_curve = auc(false_positive_rate, true_positive_rate)\n\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(false_positive_rate, true_positive_rate, label='AUC = {:.3f}'.format(area_under_curve))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.savefig('ROC_PLOT.png', bbox_inches='tight')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:54:26.971541Z","iopub.execute_input":"2021-08-19T11:54:26.971859Z","iopub.status.idle":"2021-08-19T11:54:27.790058Z","shell.execute_reply.started":"2021-08-19T11:54:26.971827Z","shell.execute_reply":"2021-08-19T11:54:27.789206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(accuracy_score(y_test, predictions))","metadata":{"execution":{"iopub.status.busy":"2021-08-19T11:54:27.791775Z","iopub.execute_input":"2021-08-19T11:54:27.792162Z","iopub.status.idle":"2021-08-19T11:54:27.798681Z","shell.execute_reply.started":"2021-08-19T11:54:27.792124Z","shell.execute_reply":"2021-08-19T11:54:27.797528Z"},"trusted":true},"execution_count":null,"outputs":[]}]}