{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install tensorflow\n!pip install keras==2.3.1\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:05:39.34457Z","iopub.execute_input":"2023-05-27T20:05:39.344938Z","iopub.status.idle":"2023-05-27T20:05:59.500804Z","shell.execute_reply.started":"2023-05-27T20:05:39.344912Z","shell.execute_reply":"2023-05-27T20:05:59.498725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd \nimport tensorflow as tf\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport tqdm\nfrom tqdm import tqdm_notebook\nfrom matplotlib.patches import Rectangle\nimport seaborn as sns\n!pip install pydicom\nimport pydicom as dcm\n%matplotlib inline \nIS_LOCAL = False\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:06:16.652935Z","iopub.execute_input":"2023-05-27T20:06:16.653209Z","iopub.status.idle":"2023-05-27T20:06:26.005293Z","shell.execute_reply.started":"2023-05-27T20:06:16.653185Z","shell.execute_reply":"2023-05-27T20:06:26.004132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detailclassinfo_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\ntrainlabels_df = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:06:26.007859Z","iopub.execute_input":"2023-05-27T20:06:26.008682Z","iopub.status.idle":"2023-05-27T20:06:26.070164Z","shell.execute_reply.started":"2023-05-27T20:06:26.008648Z","shell.execute_reply":"2023-05-27T20:06:26.068279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detailclassinfo_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:06:26.071653Z","iopub.execute_input":"2023-05-27T20:06:26.071962Z","iopub.status.idle":"2023-05-27T20:06:26.082284Z","shell.execute_reply.started":"2023-05-27T20:06:26.071942Z","shell.execute_reply":"2023-05-27T20:06:26.080839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainlabels_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:06:26.08369Z","iopub.execute_input":"2023-05-27T20:06:26.084002Z","iopub.status.idle":"2023-05-27T20:06:26.106018Z","shell.execute_reply.started":"2023-05-27T20:06:26.083976Z","shell.execute_reply":"2023-05-27T20:06:26.104765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"print('Shape of Detailed Class Information: {}'.format(detailclassinfo_df.shape))\nprint('Shape of Train Labels: {}'.format(trainlabels_df.shape))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:41.133694Z","iopub.execute_input":"2023-05-27T20:07:41.134087Z","iopub.status.idle":"2023-05-27T20:07:41.140651Z","shell.execute_reply.started":"2023-05-27T20:07:41.134056Z","shell.execute_reply":"2023-05-27T20:07:41.139002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_train_df = trainlabels_df.merge(detailclassinfo_df, left_on='patientId', right_on='patientId', how='inner')\nmerge_train_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:43.608948Z","iopub.execute_input":"2023-05-27T20:07:43.609569Z","iopub.status.idle":"2023-05-27T20:07:43.651645Z","shell.execute_reply.started":"2023-05-27T20:07:43.60954Z","shell.execute_reply":"2023-05-27T20:07:43.650554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_train_df = merge_train_df. drop_duplicates()\nmerge_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:47.268381Z","iopub.execute_input":"2023-05-27T20:07:47.268764Z","iopub.status.idle":"2023-05-27T20:07:47.290619Z","shell.execute_reply.started":"2023-05-27T20:07:47.268739Z","shell.execute_reply":"2023-05-27T20:07:47.289656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_train_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:49.867522Z","iopub.execute_input":"2023-05-27T20:07:49.868092Z","iopub.status.idle":"2023-05-27T20:07:49.884375Z","shell.execute_reply.started":"2023-05-27T20:07:49.868061Z","shell.execute_reply":"2023-05-27T20:07:49.883377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detailclassinfo_df.groupby([\"class\"]).count().transpose().style.background_gradient(cmap='Wistia',axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:52.863387Z","iopub.execute_input":"2023-05-27T20:07:52.863785Z","iopub.status.idle":"2023-05-27T20:07:52.886591Z","shell.execute_reply.started":"2023-05-27T20:07:52.863758Z","shell.execute_reply":"2023-05-27T20:07:52.88528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_train_df.groupby([\"class\"]).count().transpose().style.background_gradient(cmap='Wistia',axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:07:55.821573Z","iopub.execute_input":"2023-05-27T20:07:55.821919Z","iopub.status.idle":"2023-05-27T20:07:55.851837Z","shell.execute_reply.started":"2023-05-27T20:07:55.821899Z","shell.execute_reply":"2023-05-27T20:07:55.850515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(type(merge_train_df['class']))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:08:03.372583Z","iopub.execute_input":"2023-05-27T20:08:03.372958Z","iopub.status.idle":"2023-05-27T20:08:03.379396Z","shell.execute_reply.started":"2023-05-27T20:08:03.372931Z","shell.execute_reply":"2023-05-27T20:08:03.377554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(merge_train_df['class'].isnull().sum())\nprint(merge_train_df['class'].apply(type).value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:08:05.492821Z","iopub.execute_input":"2023-05-27T20:08:05.493203Z","iopub.status.idle":"2023-05-27T20:08:05.506844Z","shell.execute_reply.started":"2023-05-27T20:08:05.493174Z","shell.execute_reply":"2023-05-27T20:08:05.505384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade seaborn matplotlib\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:08:08.195839Z","iopub.execute_input":"2023-05-27T20:08:08.196208Z","iopub.status.idle":"2023-05-27T20:08:17.872891Z","shell.execute_reply.started":"2023-05-27T20:08:08.196181Z","shell.execute_reply":"2023-05-27T20:08:17.871519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nf, ax = plt.subplots(1, 1, figsize=(6, 4))\n\ntotal = float(len(merge_train_df))\nclass_counts = merge_train_df['class'].value_counts()\n\nsns.countplot(x='class', data=merge_train_df, order=class_counts.index, palette='Set1')\n\nfor p in ax.patches:\n    height = p.get_height()\n    ax.text(p.get_x() + p.get_width() / 2., height + 3, '{:1.2f}%'.format(100 * height / total), ha=\"center\")\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:10:43.519426Z","iopub.execute_input":"2023-05-27T20:10:43.519838Z","iopub.status.idle":"2023-05-27T20:10:43.689587Z","shell.execute_reply.started":"2023-05-27T20:10:43.519812Z","shell.execute_reply":"2023-05-27T20:10:43.687208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1,1, figsize=(6,4))\nsns.countplot(x = \"Target\", data= merge_train_df, hue=\"class\", palette='Set1')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:11:08.592738Z","iopub.execute_input":"2023-05-27T20:11:08.593063Z","iopub.status.idle":"2023-05-27T20:11:08.83402Z","shell.execute_reply.started":"2023-05-27T20:11:08.593041Z","shell.execute_reply":"2023-05-27T20:11:08.832384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('ggplot')\nf, ax = plt.subplots(1,1, figsize=(6,4))\ntotale = float(len(merge_train_df))\npd.value_counts(merge_train_df[\"Target\"]).plot(kind='bar', position=0.5, rot=0)\nfor p in ax.patches:\n    height = p.get_height()\n    ax.text(p.get_x()+p.get_width()/2.,\n            height + 3,\n            '{:1.2f}%'.format(100*height/total),\n            ha=\"center\") \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:11:21.823196Z","iopub.execute_input":"2023-05-27T20:11:21.823559Z","iopub.status.idle":"2023-05-27T20:11:21.988166Z","shell.execute_reply.started":"2023-05-27T20:11:21.823533Z","shell.execute_reply":"2023-05-27T20:11:21.987301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of positive targets\nprint(round((9555 / (9555 + 20672)) * 100, 2), '% of the patients are positive')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:11:34.679845Z","iopub.execute_input":"2023-05-27T20:11:34.680253Z","iopub.status.idle":"2023-05-27T20:11:34.685664Z","shell.execute_reply.started":"2023-05-27T20:11:34.680222Z","shell.execute_reply":"2023-05-27T20:11:34.68458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_train_df.groupby([\"Target\"]).count()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:11:46.145018Z","iopub.execute_input":"2023-05-27T20:11:46.145399Z","iopub.status.idle":"2023-05-27T20:11:46.171403Z","shell.execute_reply.started":"2023-05-27T20:11:46.145376Z","shell.execute_reply":"2023-05-27T20:11:46.169507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_bb_df = merge_train_df.drop(columns = [\"patientId\", \"class\"])\n\n## Impute NaN with KNN mean\nfrom sklearn.impute import KNNImputer\nimputer = KNNImputer(n_neighbors=2)\nmerge_bb_imputed_df = imputer.fit_transform(merge_bb_df)\n\n## Converted array to dataframe\nmerge_bb_final = pd.DataFrame(data=merge_bb_imputed_df, columns=[\"x\", \"y\", \"width\", \"height\", \"Target\"])\nmerge_bb_final[\"Target\"] = merge_bb_final[\"Target\"].astype('int64')\nmerge_bb_final.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:12:04.444993Z","iopub.execute_input":"2023-05-27T20:12:04.445361Z","iopub.status.idle":"2023-05-27T20:12:32.838179Z","shell.execute_reply.started":"2023-05-27T20:12:04.445333Z","shell.execute_reply":"2023-05-27T20:12:32.83669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seggregating Opacity data from imputed data\nopacity_bb_final = merge_bb_final[(merge_bb_final['Target']== 1)]\nopacity_bb_final.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:12:32.840779Z","iopub.execute_input":"2023-05-27T20:12:32.841317Z","iopub.status.idle":"2023-05-27T20:12:32.856289Z","shell.execute_reply.started":"2023-05-27T20:12:32.841273Z","shell.execute_reply":"2023-05-27T20:12:32.85475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(2,2,figsize=(12,12))\nsns.distplot(opacity_bb_final['x'],kde=True,bins=50, color=\"grey\", ax=ax[0,0])\nsns.distplot(opacity_bb_final['y'],kde=True,bins=50, color=\"orange\", ax=ax[0,1])\nsns.distplot(opacity_bb_final['width'],kde=True,bins=50, color=\"blue\", ax=ax[1,0])\nsns.distplot(opacity_bb_final['height'],kde=True,bins=50, color=\"brown\", ax=ax[1,1])\nlocs, labels = plt.xticks()\nplt.tick_params(axis='both', which='major', labelsize=12)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:12:32.857902Z","iopub.execute_input":"2023-05-27T20:12:32.858228Z","iopub.status.idle":"2023-05-27T20:12:34.267209Z","shell.execute_reply.started":"2023-05-27T20:12:32.8582Z","shell.execute_reply":"2023-05-27T20:12:34.266282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Correlation\nplt.figure(figsize=(7,5))\nsns.heatmap(opacity_bb_final.corr(),linewidths=0.1,vmax=1.0, \n            square=True,  linecolor='white', annot=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:12:40.275146Z","iopub.execute_input":"2023-05-27T20:12:40.275835Z","iopub.status.idle":"2023-05-27T20:12:40.550926Z","shell.execute_reply.started":"2023-05-27T20:12:40.275797Z","shell.execute_reply":"2023-05-27T20:12:40.549452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.jointplot(x = 'width', y = 'height', data = opacity_bb_final, kind= 'reg')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:13:03.415098Z","iopub.execute_input":"2023-05-27T20:13:03.415448Z","iopub.status.idle":"2023-05-27T20:13:04.817137Z","shell.execute_reply.started":"2023-05-27T20:13:03.415424Z","shell.execute_reply":"2023-05-27T20:13:04.81579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1,1,figsize=(7,7))\ntarget_sample = opacity_bb_final.sample(2000)\ntarget_sample['xc'] = target_sample['x'] + target_sample['width'] / 2\ntarget_sample['yc'] = target_sample['y'] + target_sample['height'] / 2\nplt.title(\"Lung Opacity representation for 2000 data samples\")\ntarget_sample.plot.scatter(x='xc', y='yc', xlim=(0,1024), ylim=(0,1024), ax=ax, alpha=0.8, marker=\".\", color=\"black\")\nfor i, crt_sample in target_sample.iterrows():\n    ax.add_patch(Rectangle(xy=(crt_sample['x'], crt_sample['y']),\n                width=crt_sample['width'],height=crt_sample['height'],alpha=3.5e-3, color=\"green\"))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:13:19.535062Z","iopub.execute_input":"2023-05-27T20:13:19.53546Z","iopub.status.idle":"2023-05-27T20:13:22.358967Z","shell.execute_reply.started":"2023-05-27T20:13:19.535428Z","shell.execute_reply":"2023-05-27T20:13:22.358084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = merge_train_df[[\"patientId\", \"class\"]]\ntrain_labels_data = pd.merge(tmp, merge_bb_final, how= 'inner', left_index=True, right_index=True)\ntrain_labels_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:13:34.61755Z","iopub.execute_input":"2023-05-27T20:13:34.618002Z","iopub.status.idle":"2023-05-27T20:13:34.644856Z","shell.execute_reply.started":"2023-05-27T20:13:34.61797Z","shell.execute_reply":"2023-05-27T20:13:34.643542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train_labels_data.drop(columns= [\"patientId\", \"Target\", \"class\"])\ny = train_labels_data[[\"Target\"]]\nX.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:13:44.701364Z","iopub.execute_input":"2023-05-27T20:13:44.701764Z","iopub.status.idle":"2023-05-27T20:13:44.710916Z","shell.execute_reply.started":"2023-05-27T20:13:44.701733Z","shell.execute_reply":"2023-05-27T20:13:44.70981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade scikit-learn\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:14:27.718351Z","iopub.execute_input":"2023-05-27T20:14:27.718753Z","iopub.status.idle":"2023-05-27T20:14:37.032209Z","shell.execute_reply.started":"2023-05-27T20:14:27.718726Z","shell.execute_reply":"2023-05-27T20:14:37.030809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the confusion matrix\ncm = confusion_matrix(y_test, y_pred)\n\n# Plot the confusion matrix\nplt.figure(figsize=(8, 6))\nsns.heatmap(cm, annot=True, cmap=\"Blues\", fmt=\"d\", cbar=False)\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted Labels\")\nplt.ylabel(\"True Labels\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:13.212103Z","iopub.execute_input":"2023-05-27T20:22:13.212502Z","iopub.status.idle":"2023-05-27T20:22:13.356322Z","shell.execute_reply.started":"2023-05-27T20:22:13.212456Z","shell.execute_reply":"2023-05-27T20:22:13.355662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_curve, auc\nimport matplotlib.pyplot as plt\n\n# Calculate the false positive rate (FPR), true positive rate (TPR), and thresholds\nfpr, tpr, thresholds = roc_curve(y_test, y_pred)\n\n# Calculate the Area Under the Curve (AUC)\nroc_auc = auc(fpr, tpr)\n\n# Plot the ROC curve\nplt.figure(figsize=(8, 6))\nplt.plot(fpr, tpr, label='ROC curve (AUC = %0.2f)' % roc_auc)\nplt.plot([0, 1], [0, 1], 'k--')  # Random guessing curve\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate (FPR)')\nplt.ylabel('True Positive Rate (TPR)')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc=\"lower right\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:17.316155Z","iopub.execute_input":"2023-05-27T20:22:17.316477Z","iopub.status.idle":"2023-05-27T20:22:17.544414Z","shell.execute_reply.started":"2023-05-27T20:22:17.316456Z","shell.execute_reply":"2023-05-27T20:22:17.543568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nsvm_model = SVC(kernel = 'rbf', C = 1.0).fit(X_train, y_train)\ny_preds = svm_model.predict(X_test)\nprint('Accuracy score:', accuracy_score(y_test, y_preds))\nprint('Average Precision Score:', average_precision_score(y_test, y_preds))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:21.883505Z","iopub.execute_input":"2023-05-27T20:22:21.88387Z","iopub.status.idle":"2023-05-27T20:22:21.959279Z","shell.execute_reply.started":"2023-05-27T20:22:21.883843Z","shell.execute_reply":"2023-05-27T20:22:21.95827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Score: \"{:.2%}\"'.format(svm_model.score(X, y)))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:24.588125Z","iopub.execute_input":"2023-05-27T20:22:24.588468Z","iopub.status.idle":"2023-05-27T20:22:24.668406Z","shell.execute_reply.started":"2023-05-27T20:22:24.58844Z","shell.execute_reply":"2023-05-27T20:22:24.667538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Classification report: \\n')\nprint(classification_report(y_test, y_preds))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:30.710246Z","iopub.execute_input":"2023-05-27T20:22:30.710959Z","iopub.status.idle":"2023-05-27T20:22:30.732136Z","shell.execute_reply.started":"2023-05-27T20:22:30.710909Z","shell.execute_reply":"2023-05-27T20:22:30.730677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport matplotlib.pyplot as plt\n\n# Calculate the confusion matrix\ncm = confusion_matrix(y_test, y_pred)\n\n# Create the ConfusionMatrixDisplay object\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)\n\n# Plot the confusion matrix\nfig, ax = plt.subplots(figsize=(8, 6))\ndisp.plot(cmap='plasma', ax=ax)\nplt.title('Confusion Matrix')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:33.583792Z","iopub.execute_input":"2023-05-27T20:22:33.584151Z","iopub.status.idle":"2023-05-27T20:22:33.805063Z","shell.execute_reply.started":"2023-05-27T20:22:33.584124Z","shell.execute_reply":"2023-05-27T20:22:33.804237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fpr, tpr, thresholds = roc_curve(y_test, y_preds)\n\n# calculate AUC\nauc = roc_auc_score(y_test, y_preds)\nprint('AUC: %.3f' % auc)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:22:47.624913Z","iopub.execute_input":"2023-05-27T20:22:47.625247Z","iopub.status.idle":"2023-05-27T20:22:47.647064Z","shell.execute_reply.started":"2023-05-27T20:22:47.625223Z","shell.execute_reply":"2023-05-27T20:22:47.645411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patientId = trainlabels_df['patientId'][0]\ndcm_file = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % patientId\ndcm_data = dcm.read_file(dcm_file)\nprint(dcm_data)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:23:00.598938Z","iopub.execute_input":"2023-05-27T20:23:00.599327Z","iopub.status.idle":"2023-05-27T20:23:00.627221Z","shell.execute_reply.started":"2023-05-27T20:23:00.5993Z","shell.execute_reply":"2023-05-27T20:23:00.62604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"im = dcm_data.pixel_array\nprint(type(im))\nprint(im.dtype)\nprint(im.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:23:10.063813Z","iopub.execute_input":"2023-05-27T20:23:10.064143Z","iopub.status.idle":"2023-05-27T20:23:10.078389Z","shell.execute_reply.started":"2023-05-27T20:23:10.06412Z","shell.execute_reply":"2023-05-27T20:23:10.077307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pylab\npylab.imshow(im, cmap=pylab.cm.gist_gray)\npylab.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:23:19.475151Z","iopub.execute_input":"2023-05-27T20:23:19.475522Z","iopub.status.idle":"2023-05-27T20:23:19.650295Z","shell.execute_reply.started":"2023-05-27T20:23:19.475475Z","shell.execute_reply":"2023-05-27T20:23:19.649453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_data(df):\n    \"\"\"\n    Method to read a CSV file (Pandas dataframe) and parse the \n    data into the following nested dictionary:\n\n      parsed = {\n        \n        'patientId-00': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        },\n        'patientId-01': {\n            'dicom': path/to/dicom/file,\n            'label': either 0 or 1 for normal or pnuemonia, \n            'boxes': list of box(es)\n        }, ...\n\n      }\n\n    \"\"\"\n    # --- Define lambda to extract coords in list [y, x, height, width]\n    extract_box = lambda row: [row['y'], row['x'], row['height'], row['width']]\n\n    parsed = {}\n    for n, row in df.iterrows():\n        # --- Initialize patient entry into parsed \n        pid = row['patientId']\n        if pid not in parsed:\n            parsed[pid] = {\n                'dicom': '../input/rsna-pneumonia-detection-challenge/stage_2_train_images/%s.dcm' % pid,\n                'label': row['Target'],\n                'boxes': []}\n\n        # --- Add box if opacity is present\n        if parsed[pid]['label'] == 1:\n            parsed[pid]['boxes'].append(extract_box(row))\n\n    return parsed\n\nparsed = parse_data(trainlabels_df)\n\ndef draw(data):\n    \"\"\"\n    Method to draw single patient with bounding box(es) if present \n\n    \"\"\"\n    # --- Open DICOM file\n    d = dcm.read_file(data['dicom'])\n    im = d.pixel_array\n\n    # --- Convert from single-channel grayscale to 3-channel RGB\n    im = np.stack([im] * 3, axis=2)\n\n    # --- Add boxes with random color if present\n    for box in data['boxes']:\n        #rgb = np.floor(np.random.rand(3) * 256).astype('int')\n        rgb = [0, 0, 255] # Just use blue\n        im = overlay_box(im=im, box=box, rgb=rgb, stroke=15)\n\n    plt.imshow(im, cmap=plt.cm.gist_gray)\n    plt.axis('off')\n\ndef overlay_box(im, box, rgb, stroke=2):\n    \"\"\"\n    Method to overlay single box on image\n\n    \"\"\"\n    # --- Convert coordinates to integers\n    box = [int(b) for b in box]\n    \n    # --- Extract coordinates\n    y1, x1, height, width = box\n    y2 = y1 + height\n    x2 = x1 + width\n\n    im[y1:y1 + stroke, x1:x2] = rgb\n    im[y2:y2 + stroke, x1:x2] = rgb\n    im[y1:y2, x1:x1 + stroke] = rgb\n    im[y1:y2, x2:x2 + stroke] = rgb\n\n    return im","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:23:47.345776Z","iopub.execute_input":"2023-05-27T20:23:47.346159Z","iopub.status.idle":"2023-05-27T20:23:48.884326Z","shell.execute_reply.started":"2023-05-27T20:23:47.346131Z","shell.execute_reply":"2023-05-27T20:23:48.883319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\nfig=plt.figure(figsize=(8, 8))\ncolumns = 5; rows = 4\nfor i in range(1, columns*rows +1):\n    fig.add_subplot(rows, columns, i)\n    draw(parsed[trainlabels_df['patientId'].unique()[i]])\n    fig.add_subplot","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:24:05.095599Z","iopub.execute_input":"2023-05-27T20:24:05.095938Z","iopub.status.idle":"2023-05-27T20:24:09.235172Z","shell.execute_reply.started":"2023-05-27T20:24:05.095916Z","shell.execute_reply":"2023-05-27T20:24:09.234013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opacity = detailclassinfo_df \\\n    .loc[detailclassinfo_df['class'] == 'Lung Opacity'] \\\n    .reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:24:16.446701Z","iopub.execute_input":"2023-05-27T20:24:16.447064Z","iopub.status.idle":"2023-05-27T20:24:16.455557Z","shell.execute_reply.started":"2023-05-27T20:24:16.447037Z","shell.execute_reply":"2023-05-27T20:24:16.454926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"not_normal = detailclassinfo_df \\\n    .loc[detailclassinfo_df['class'] == 'No Lung Opacity / Not Normal'] \\\n    .reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:24:29.432945Z","iopub.execute_input":"2023-05-27T20:24:29.433334Z","iopub.status.idle":"2023-05-27T20:24:29.443236Z","shell.execute_reply.started":"2023-05-27T20:24:29.433305Z","shell.execute_reply":"2023-05-27T20:24:29.44214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"normal = detailclassinfo_df \\\n    .loc[detailclassinfo_df['class'] == 'Normal'] \\\n    .reset_index()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:24:39.5777Z","iopub.execute_input":"2023-05-27T20:24:39.578029Z","iopub.status.idle":"2023-05-27T20:24:39.587599Z","shell.execute_reply.started":"2023-05-27T20:24:39.578007Z","shell.execute_reply":"2023-05-27T20:24:39.586341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\nfig=plt.figure(figsize=(6, 6))\ncolumns = 4; rows = 4\nfor i in range(1, columns*rows +1):\n    fig.add_subplot(rows, columns, i)\n    draw(parsed[opacity['patientId'].unique()[i]])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:24:51.492662Z","iopub.execute_input":"2023-05-27T20:24:51.493076Z","iopub.status.idle":"2023-05-27T20:24:54.789508Z","shell.execute_reply.started":"2023-05-27T20:24:51.49304Z","shell.execute_reply":"2023-05-27T20:24:54.788334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\nfig=plt.figure(figsize=(6, 6))\ncolumns = 4; rows = 4\nfor i in range(1, columns*rows +1):\n    fig.add_subplot(rows, columns, i)\n    draw(parsed[normal['patientId'].unique()[i]])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:25:20.096217Z","iopub.execute_input":"2023-05-27T20:25:20.096611Z","iopub.status.idle":"2023-05-27T20:25:23.786794Z","shell.execute_reply.started":"2023-05-27T20:25:20.096583Z","shell.execute_reply":"2023-05-27T20:25:23.78514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('default')\nfig=plt.figure(figsize=(6, 6))\ncolumns = 4; rows = 4\nfor i in range(1, columns*rows +1):\n    fig.add_subplot(rows, columns, i)\n    draw(parsed[not_normal['patientId'].unique()[i]])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:25:31.421767Z","iopub.execute_input":"2023-05-27T20:25:31.422663Z","iopub.status.idle":"2023-05-27T20:25:34.699996Z","shell.execute_reply.started":"2023-05-27T20:25:31.42263Z","shell.execute_reply":"2023-05-27T20:25:34.698768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig=plt.figure(figsize=(20, 10))\ncolumns = 3; rows = 1\nfig.add_subplot(rows, columns, 1).set_title(\"Normal\", fontsize=30)\ndraw(parsed[normal['patientId'].unique()[0]])\nfig.add_subplot(rows, columns, 2).set_title(\"Not Normal\", fontsize=30)\n# ax2.set_title(\"Not Normal\", fontsize=30)\ndraw(parsed[not_normal['patientId'].unique()[0]])\nfig.add_subplot(rows, columns, 3).set_title(\"Opacity\", fontsize=30)\n# ax3.set_title(\"Opacity\", fontsize=30)\ndraw(parsed[opacity['patientId'].unique()[0]])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:25:44.209579Z","iopub.execute_input":"2023-05-27T20:25:44.209973Z","iopub.status.idle":"2023-05-27T20:25:45.271113Z","shell.execute_reply.started":"2023-05-27T20:25:44.209945Z","shell.execute_reply":"2023-05-27T20:25:45.270378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Image Segmentation**\n\nImage segmentation creates a pixel-wise mask for each object in the image. This technique gives us a far more granular understanding of the object(s) in the image.\n\nClass generatortransfer:\n\nThe dataset is too large to fit into memory, so we need to create a generator that loads data on the fly.\n\nI/p: filenames, batch_size and other parameters.\n\nO/p: A random batch of numpy images and numpy masks.","metadata":{}},{"cell_type":"markdown","source":"**Model Selection**","metadata":{}},{"cell_type":"code","source":"import os\nimport csv\nimport random\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom skimage import measure\nfrom skimage.transform import resize\n\nimport tensorflow as tf\nfrom tensorflow import keras","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:27:21.059702Z","iopub.execute_input":"2023-05-27T20:27:21.060081Z","iopub.status.idle":"2023-05-27T20:27:21.291228Z","shell.execute_reply.started":"2023-05-27T20:27:21.060052Z","shell.execute_reply":"2023-05-27T20:27:21.289804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"opacity_locations = {}\n# load table\nwith open(os.path.join('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv'), mode='r') as infile:\n    # open reader\n    reader = csv.reader(infile)\n    # skip header\n    next(reader, None)\n    # loop through rows\n    for rows in reader:\n        # retrieve information\n        filename = rows[0]\n        location = rows[1:5]\n        lungopacity = rows[5]\n        # if row contains lungopacity add label to dictionary\n        # which contains a list of lungopacity locations per filename\n        if lungopacity == '1':\n            # convert string to float to int\n            location = [int(float(i)) for i in location]\n            # save lungopacity location in dictionary\n            if filename in opacity_locations:\n                opacity_locations[filename].append(location)\n            else:\n                opacity_locations[filename] = [location]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:27:25.891197Z","iopub.execute_input":"2023-05-27T20:27:25.891568Z","iopub.status.idle":"2023-05-27T20:27:25.954087Z","shell.execute_reply.started":"2023-05-27T20:27:25.891538Z","shell.execute_reply":"2023-05-27T20:27:25.953019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\nfilenames = os.listdir(folder)\nprint('Number of Train images:', len(filenames))\nprint('First 5 samples: \\n')\nfilenames[0:5]","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:27:39.587744Z","iopub.execute_input":"2023-05-27T20:27:39.588128Z","iopub.status.idle":"2023-05-27T20:27:39.606384Z","shell.execute_reply.started":"2023-05-27T20:27:39.588099Z","shell.execute_reply":"2023-05-27T20:27:39.605423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#folder = '/content/drive/My Drive/GL_Capstone/stage_2_train_images'\n#filenames = os.listdir(folder)\nrandom.shuffle(filenames)\n# split into train and validation filenames\nn_valid_samples = 5336\nn_train_samples = len(filenames) - n_valid_samples\ntrain_filenames = filenames[n_valid_samples:]\nvalid_filenames = filenames[:n_valid_samples]\nprint('Total file samples:', len(filenames))\nprint('Train samples (80%):', len(train_filenames))\nprint('Valid samples (20%):', len(valid_filenames))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:27:50.317679Z","iopub.execute_input":"2023-05-27T20:27:50.318068Z","iopub.status.idle":"2023-05-27T20:27:50.344195Z","shell.execute_reply.started":"2023-05-27T20:27:50.318039Z","shell.execute_reply":"2023-05-27T20:27:50.342976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_downsamples = 5336\ntrain_dsamps = train_filenames[:n_downsamples]\nprint('Training Downsamples:', len(train_dsamps))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:28:01.630968Z","iopub.execute_input":"2023-05-27T20:28:01.631338Z","iopub.status.idle":"2023-05-27T20:28:01.637293Z","shell.execute_reply.started":"2023-05-27T20:28:01.631311Z","shell.execute_reply":"2023-05-27T20:28:01.636271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install opencv-python\nimport cv2\nkeras = tf.compat.v1.keras\n#Sequence = keras.utils.Sequence","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:28:12.32024Z","iopub.execute_input":"2023-05-27T20:28:12.320614Z","iopub.status.idle":"2023-05-27T20:28:21.526808Z","shell.execute_reply.started":"2023-05-27T20:28:12.320588Z","shell.execute_reply":"2023-05-27T20:28:21.525147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define class generatortransfer","metadata":{}},{"cell_type":"code","source":"pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:34:29.25337Z","iopub.execute_input":"2023-05-27T20:34:29.254802Z","iopub.status.idle":"2023-05-27T20:34:39.66132Z","shell.execute_reply.started":"2023-05-27T20:34:29.254745Z","shell.execute_reply":"2023-05-27T20:34:39.659753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow-gpu","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:34:43.010521Z","iopub.execute_input":"2023-05-27T20:34:43.010924Z","iopub.status.idle":"2023-05-27T20:34:46.347138Z","shell.execute_reply.started":"2023-05-27T20:34:43.010893Z","shell.execute_reply":"2023-05-27T20:34:46.34594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Use tf.keras instead of keras\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:34:50.990697Z","iopub.execute_input":"2023-05-27T20:34:50.991106Z","iopub.status.idle":"2023-05-27T20:34:50.99928Z","shell.execute_reply.started":"2023-05-27T20:34:50.991074Z","shell.execute_reply":"2023-05-27T20:34:50.998508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class generatortransfer(keras.utils.Sequence):\n    \n    def __init__(self, folder, filenames, opacity_locations=None, batch_size=32, image_size=320, shuffle=True, augment=False, predict=False):\n        self.folder = folder\n        self.filenames = filenames\n        self.opacity_locations = opacity_locations\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.predict = predict\n        self.on_epoch_end()\n        \n    def __load__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # create empty mask\n        msk = np.zeros(img.shape)\n        # get filename without extension\n        filename = filename.split('.')[0]\n        # if image contains lung opacity\n        if filename in opacity_locations:\n            # loop through opacity\n            for location in opacity_locations[filename]:\n                # add 1's at the location of the lung opacity\n                x, y, w, h = location\n                msk[y:y+h, x:x+w] = 1\n        # if augment then horizontal flip half the time\n        if self.augment and random.random() > 0.5:\n            img = np.fliplr(img)\n            msk = np.fliplr(msk)\n        # resize both image and mask\n        #img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        msk = resize(msk, (self.image_size, self.image_size), mode='reflect') > 0.5\n        # add trailing channel dimension\n        msk = np.expand_dims(msk, -1)\n         #Converting Image from GrayScale to RGB \n        if len(img.shape) != 3 or img.shape[2] != 3:\n            img = np.stack((img,) * 3, -1)\n            img = cv2.resize(img, dsize=(self.image_size, self.image_size), interpolation=cv2.INTER_CUBIC)\n        return img, msk\n    \n    def __loadpredict__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # resize image\n        #img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        #Converting Image from GrayScale to RGB \n        if len(img.shape) != 3 or img.shape[2] != 3:\n          img = np.stack((img,) * 3, -1)\n          img = cv2.resize(img, dsize=(self.image_size, self.image_size), interpolation=cv2.INTER_CUBIC)\n        return img\n        \n    def __getitem__(self, index):\n        # select batch\n        filenames = self.filenames[index*self.batch_size:(index+1)*self.batch_size]\n        # predict mode: return images and filenames\n        if self.predict:\n            # load files\n            imgs = [self.__loadpredict__(filename) for filename in filenames]\n            # create numpy batch\n            imgs = np.array(imgs)\n            return imgs, filenames\n        # train mode: return images and masks\n        else:\n            # load files\n            items = [self.__load__(filename) for filename in filenames]\n            # unzip images and masks\n            imgs, msks = zip(*items)\n            # create numpy batch\n            imgs = np.array(imgs)\n            msks = np.array(msks)\n            return imgs, msks\n        \n    def on_epoch_end(self):\n        if self.shuffle:\n            random.shuffle(self.filenames)\n        \n    def __len__(self):\n        if self.predict:\n            # return everything\n            return int(np.ceil(len(self.filenames) / self.batch_size))\n        else:\n            # return full batches only\n            return int(len(self.filenames) / self.batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:34:54.363389Z","iopub.execute_input":"2023-05-27T20:34:54.363771Z","iopub.status.idle":"2023-05-27T20:34:54.406933Z","shell.execute_reply.started":"2023-05-27T20:34:54.363743Z","shell.execute_reply":"2023-05-27T20:34:54.405781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_width = 128\nimg_height = 128\nIMAGE_SIZE=128\nkernel =3\nnum_of_classes =2\nBATCH_SIZE = 16\nSHUFFLE_BUFFER_SIZE=1000\n#input_shape = (img_width, img_height, 3)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:35:07.910225Z","iopub.execute_input":"2023-05-27T20:35:07.910583Z","iopub.status.idle":"2023-05-27T20:35:07.916802Z","shell.execute_reply.started":"2023-05-27T20:35:07.910559Z","shell.execute_reply":"2023-05-27T20:35:07.915498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create training and validation data\nfolder = '../input/rsna-pneumonia-detection-challenge/stage_2_train_images'\ntrain_trans = generatortransfer(folder, train_filenames, opacity_locations, batch_size=BATCH_SIZE, image_size=IMAGE_SIZE, shuffle=True, augment=False, predict=False)\nvalid_trans = generatortransfer(folder, valid_filenames, opacity_locations, batch_size=BATCH_SIZE, image_size=IMAGE_SIZE, shuffle=False, predict=False)\n#train_downsamples = generatortransfer(folder, tune_train_samples, opacity_locations, batch_size=BATCH_SIZE, image_size=IMAGE_SIZE, shuffle=False, predict=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:35:20.782703Z","iopub.execute_input":"2023-05-27T20:35:20.783074Z","iopub.status.idle":"2023-05-27T20:35:20.806973Z","shell.execute_reply.started":"2023-05-27T20:35:20.783043Z","shell.execute_reply":"2023-05-27T20:35:20.805156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom tensorflow.keras import Sequential, backend as K\n#from tensorflow.keras.applications.mobilenet import MobileNet\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.layers import Concatenate, UpSampling2D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense\n\nmodel = Sequential()\nmodel.add(ResNet50(input_shape= (img_width, img_height, 3), include_top=False, weights='imagenet'))\nmodel.add(Dense(1024, activation='relu'))\nmodel.add(UpSampling2D())\nmodel.add(Dense(512, activation='relu'))\nmodel.add(UpSampling2D())\nmodel.add(Dense(256, activation='relu'))\nmodel.add(UpSampling2D())\nmodel.add(Dense(64, activation='relu'))\nmodel.add(UpSampling2D())\nmodel.add(Dense(8, activation='relu'))\nmodel.add(UpSampling2D())\nmodel.add(Dense(1, activation='sigmoid'))\n# Say not to train first layer (ResNet) model. It is already trained\nmodel.layers[0].trainable = False\nprint(model.summary())","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:35:34.366283Z","iopub.execute_input":"2023-05-27T20:35:34.366616Z","iopub.status.idle":"2023-05-27T20:35:40.170871Z","shell.execute_reply.started":"2023-05-27T20:35:34.366594Z","shell.execute_reply":"2023-05-27T20:35:40.169654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='Adam',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:35:49.786021Z","iopub.execute_input":"2023-05-27T20:35:49.786365Z","iopub.status.idle":"2023-05-27T20:35:49.808281Z","shell.execute_reply.started":"2023-05-27T20:35:49.78634Z","shell.execute_reply":"2023-05-27T20:35:49.806942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(train_trans, epochs=5, steps_per_epoch =10, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:36:01.523022Z","iopub.execute_input":"2023-05-27T20:36:01.523393Z","iopub.status.idle":"2023-05-27T20:37:13.781597Z","shell.execute_reply.started":"2023-05-27T20:36:01.523364Z","shell.execute_reply":"2023-05-27T20:37:13.780759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(valid_trans)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:37:13.783436Z","iopub.execute_input":"2023-05-27T20:37:13.783786Z","iopub.status.idle":"2023-05-27T20:37:28.51185Z","shell.execute_reply.started":"2023-05-27T20:37:13.783762Z","shell.execute_reply":"2023-05-27T20:37:28.509969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Hyper Parameter Tuning**\nI have tried couple of times to use GridSearchCV in order to find out the optimal parameters, but the session ran for longer time and the runtime was disconnected and finally lead to lost of changes.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:37:28.512749Z","iopub.status.idle":"2023-05-27T20:37:28.51309Z","shell.execute_reply.started":"2023-05-27T20:37:28.512931Z","shell.execute_reply":"2023-05-27T20:37:28.512945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow==1.14.0\n!pip install keras==2.0","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:37:31.353208Z","iopub.execute_input":"2023-05-27T20:37:31.353584Z","iopub.status.idle":"2023-05-27T20:39:59.490413Z","shell.execute_reply.started":"2023-05-27T20:37:31.353556Z","shell.execute_reply":"2023-05-27T20:39:59.488041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mean iou as a metric\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2, 3])\n    union = tf.reduce_sum(y_true, axis=[1, 2, 3]) + tf.reduce_sum(y_pred, axis=[1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:39:59.494695Z","iopub.execute_input":"2023-05-27T20:39:59.495158Z","iopub.status.idle":"2023-05-27T20:39:59.504535Z","shell.execute_reply.started":"2023-05-27T20:39:59.495124Z","shell.execute_reply":"2023-05-27T20:39:59.503085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Call Backs (Earlystop, ModelCheckpoint)","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\n## Earlystopping\nearlystop = EarlyStopping(monitor='val_loss', patience=3)\n\n## Model Check point\nfilepath=\"/kaggle/working/LOHPT-{epoch:02d}-{val_accuracy:.2f}.hdf5\"\ncheckpoint = ModelCheckpoint(filepath, monitor='val_loss', verbose=1, save_best_only=True, mode='min', save_weights_only = True)\n\n## Reduce learning rate when metric has stopped improving\nreduceLROnPlat = ReduceLROnPlateau(monitor='val_loss', factor=0.8, \n                                   patience=2, verbose=1, mode='auto', \n                                   epsilon=0.0001, cooldown=5, min_lr=0.0001)","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:39:59.505899Z","iopub.execute_input":"2023-05-27T20:39:59.506883Z","iopub.status.idle":"2023-05-27T20:39:59.532064Z","shell.execute_reply.started":"2023-05-27T20:39:59.506847Z","shell.execute_reply":"2023-05-27T20:39:59.530667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define Parameters","metadata":{}},{"cell_type":"code","source":"BATCH_SIZE = 16\nIMAGE_SIZE = 128","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:40:04.937166Z","iopub.execute_input":"2023-05-27T20:40:04.937675Z","iopub.status.idle":"2023-05-27T20:40:04.943796Z","shell.execute_reply.started":"2023-05-27T20:40:04.937625Z","shell.execute_reply":"2023-05-27T20:40:04.941728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport keras\nfrom tensorflow.keras import Sequential, backend as K\n#from tensorflow.keras.applications.mobilenet import MobileNet\nfrom tensorflow.keras.applications.resnet50 import ResNet50\nfrom tensorflow.keras.layers import Concatenate, UpSampling2D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense\n\nhpt_model = Sequential()\nhpt_model.add(ResNet50(input_shape= (img_width, img_height, 3), include_top=False, weights='imagenet'))\nhpt_model.add(Dense(1024, activation='relu'))\nhpt_model.add(UpSampling2D())\nhpt_model.add(Dense(512, activation='relu'))\nhpt_model.add(UpSampling2D())\nhpt_model.add(Dense(256, activation='relu'))\nhpt_model.add(UpSampling2D())\nhpt_model.add(Dense(64, activation='relu'))\nhpt_model.add(UpSampling2D())\nhpt_model.add(Dense(8, activation='relu'))\nhpt_model.add(UpSampling2D())\nhpt_model.add(Dense(1, activation='sigmoid'))\n# Say not to train first layer (ResNet) model. It is already trained\nhpt_model.layers[0].trainable = False\nprint(hpt_model.summary())","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:39:59.534919Z","iopub.execute_input":"2023-05-27T20:39:59.535566Z","iopub.status.idle":"2023-05-27T20:40:04.934733Z","shell.execute_reply.started":"2023-05-27T20:39:59.535528Z","shell.execute_reply":"2023-05-27T20:40:04.932881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Compilation","metadata":{}},{"cell_type":"code","source":"hpt_model.compile(optimizer= 'Adam',\n              loss='binary_crossentropy',\n              metrics=['accuracy', mean_iou])","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:40:04.945953Z","iopub.execute_input":"2023-05-27T20:40:04.946374Z","iopub.status.idle":"2023-05-27T20:40:04.975552Z","shell.execute_reply.started":"2023-05-27T20:40:04.94634Z","shell.execute_reply":"2023-05-27T20:40:04.974282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Fitting Model","metadata":{}},{"cell_type":"code","source":"history = hpt_model.fit(train_trans, validation_data=valid_trans)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:42:24.007379Z","iopub.execute_input":"2023-05-27T20:42:24.007836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plots for Accuracy, Loss, IoU Mean","metadata":{}},{"cell_type":"markdown","source":"Accuracy","metadata":{}},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nepochs = range(1, len(acc) + 1)\nplt.plot(epochs, acc, 'y', label='Training Accuracy')\nplt.plot(epochs, val_acc, 'r', label='Validation Accuracy')\nplt.title('Training and Validation Accuracy Graph')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-27T20:40:53.255644Z","iopub.execute_input":"2023-05-27T20:40:53.256098Z","iopub.status.idle":"2023-05-27T20:40:53.306151Z","shell.execute_reply.started":"2023-05-27T20:40:53.256048Z","shell.execute_reply":"2023-05-27T20:40:53.304631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loss","metadata":{}},{"cell_type":"code","source":"Train_Loss = history.history['loss']\nVal_Loss = history.history['val_loss']\nEpochs = range(1, len(Train_Loss) + 1)\nplt.plot(Epochs, Train_Loss, 'r', label='Training Loss')\nplt.plot(Epochs, Val_Loss, 'g', label='Validation Loss')\nplt.title('Training and Validation Loss Graph')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"iou_mean","metadata":{}},{"cell_type":"code","source":"train_iou = history.history['mean_iou']\nval_iou = history.history['val_mean_iou']\nepochs = range(1, len(train_iou) + 1)\nplt.plot(epochs, train_iou, 'r', label='Training iou')\nplt.plot(epochs, val_iou, 'g', label='Validation iou')\nplt.title('Training and Validation IOU Graph')\nplt.xlabel('Epochs')\nplt.ylabel('IOU')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load the model weights","metadata":{}},{"cell_type":"code","source":"model.save_weights('/kaggle/working/RESNET50-05-0.97.hdf5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Predict Test images","metadata":{}},{"cell_type":"code","source":"# load and shuffle filenames\n#folder = '../input/stage_2_test_images'\nfolder = '../input/rsna-pneumonia-detection-challenge/stage_2_test_images'\ntest_filenames = os.listdir(folder)\nprint('n test samples:', len(test_filenames))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create test generator with predict flag set to True\ntest_trans = generatortransfer(folder, test_filenames, None, batch_size=16, image_size=IMAGE_SIZE, shuffle=False, predict=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dicom_dir = '../input/rsna-pneumonia-detection-challenge/stage_2_test_images'","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\ndef get_dicom_fps(dicom_dir):\n    dicom_fps = glob.glob(dicom_dir+'/'+'*.dcm')\n    return list(set(dicom_fps))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get filenames of test dataset DICOM images\ntest_image_fps = get_dicom_fps(test_dicom_dir)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on test images, write out sample submission \ndef predict(image_fps, filepath='/content/sample_submission.csv', min_conf=0.98): \n    \n    # assume square image\n    \n    with open(filepath, 'w') as file:\n      for image_id in tqdm_notebook(image_fps): \n        ds = dcm.read_file(image_id)\n        image = ds.pixel_array\n          \n        # If grayscale. Convert to RGB for consistency.\n        #if len(image.shape) != 3 or image.shape[2] != 3:\n\n        image = np.stack((image,) * 3, -1)\n        #img = cv2.resize(image, dsize=(128, 128), interpolation=cv2.INTER_CUBIC)\n\n\n        #patient_id = os.path.splitext(os.path.basename(image_fps))[0]\n        patient_id = image_fps\n        results = hpt_model.predict([image])\n        r = results[0]\n\n        out_str = \"\"\n        out_str += patient_id \n        assert( len(r['rois']) == len(r['class_ids']) == len(r['scores']) )\n        if len(r['rois']) == 0: \n            pass\n        else: \n            num_instances = len(r['rois'])\n            out_str += \",\"\n            for i in range(num_instances): \n                if r['scores'][i] > min_conf: \n                    out_str += ' '\n                    out_str += str(round(r['scores'][i], 2))\n                    out_str += ' '\n\n                    # x1, y1, width, height \n                    x1 = r['rois'][i][1]\n                    y1 = r['rois'][i][0]\n                    width = r['rois'][i][3] - x1 \n                    height = r['rois'][i][2] - y1 \n                    bboxes_str = \"{} {} {} {}\".format(x1, y1, \\\n                                                      width, height)    \n                    out_str += bboxes_str\n\n        filepath.write(out_str+\"\\n\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict only the first 50 entries\nsample_submission_fp = '/kaggle/working/sample_submission.csv'\npredict(test_image_fps[:50], filepath=sample_submission_fp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}