{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\n\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\n\nimport matplotlib.pyplot as plt\nimport matplotlib.style  as style\n\nfrom tqdm  import tqdm\nfrom sklearn.metrics     import accuracy_score, roc_auc_score\nfrom sklearn.linear_model      import LogisticRegression\nfrom sklearn.model_selection   import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-24T18:27:03.384812Z","iopub.execute_input":"2021-07-24T18:27:03.385360Z","iopub.status.idle":"2021-07-24T18:27:04.444389Z","shell.execute_reply.started":"2021-07-24T18:27:03.385254Z","shell.execute_reply":"2021-07-24T18:27:04.443421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn  -U","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:04.445651Z","iopub.execute_input":"2021-07-24T18:27:04.445890Z","iopub.status.idle":"2021-07-24T18:27:18.172821Z","shell.execute_reply.started":"2021-07-24T18:27:04.445867Z","shell.execute_reply":"2021-07-24T18:27:18.171888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/train_image_level.csv')","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.174544Z","iopub.execute_input":"2021-07-24T18:27:18.174892Z","iopub.status.idle":"2021-07-24T18:27:18.220679Z","shell.execute_reply.started":"2021-07-24T18:27:18.174863Z","shell.execute_reply":"2021-07-24T18:27:18.219833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.222286Z","iopub.execute_input":"2021-07-24T18:27:18.222657Z","iopub.status.idle":"2021-07-24T18:27:18.230254Z","shell.execute_reply.started":"2021-07-24T18:27:18.222604Z","shell.execute_reply":"2021-07-24T18:27:18.229393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.231081Z","iopub.execute_input":"2021-07-24T18:27:18.231398Z","iopub.status.idle":"2021-07-24T18:27:18.254035Z","shell.execute_reply.started":"2021-07-24T18:27:18.231373Z","shell.execute_reply":"2021-07-24T18:27:18.253428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"] = df.apply(lambda x: \"none\" if \"none\" in x[\"label\"] else \"opacity\", axis = 1)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.255096Z","iopub.execute_input":"2021-07-24T18:27:18.255531Z","iopub.status.idle":"2021-07-24T18:27:18.331006Z","shell.execute_reply.started":"2021-07-24T18:27:18.255502Z","shell.execute_reply":"2021-07-24T18:27:18.330423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(df.columns[1:4], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.332060Z","iopub.execute_input":"2021-07-24T18:27:18.332430Z","iopub.status.idle":"2021-07-24T18:27:18.337983Z","shell.execute_reply.started":"2021-07-24T18:27:18.332404Z","shell.execute_reply":"2021-07-24T18:27:18.337161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.340095Z","iopub.execute_input":"2021-07-24T18:27:18.340352Z","iopub.status.idle":"2021-07-24T18:27:18.351256Z","shell.execute_reply.started":"2021-07-24T18:27:18.340330Z","shell.execute_reply":"2021-07-24T18:27:18.350624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"].unique()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.352782Z","iopub.execute_input":"2021-07-24T18:27:18.353035Z","iopub.status.idle":"2021-07-24T18:27:18.359515Z","shell.execute_reply.started":"2021-07-24T18:27:18.353012Z","shell.execute_reply":"2021-07-24T18:27:18.358585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_class_to_num = {\"none\":0, \"opacity\":1}","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.360800Z","iopub.execute_input":"2021-07-24T18:27:18.361055Z","iopub.status.idle":"2021-07-24T18:27:18.367477Z","shell.execute_reply.started":"2021-07-24T18:27:18.361031Z","shell.execute_reply":"2021-07-24T18:27:18.366683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_class_to_num","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.368827Z","iopub.execute_input":"2021-07-24T18:27:18.369391Z","iopub.status.idle":"2021-07-24T18:27:18.376440Z","shell.execute_reply.started":"2021-07-24T18:27:18.369349Z","shell.execute_reply":"2021-07-24T18:27:18.375762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"] = df[\"study_class\"].apply(lambda x: study_class_to_num[x])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.377369Z","iopub.execute_input":"2021-07-24T18:27:18.377639Z","iopub.status.idle":"2021-07-24T18:27:18.387412Z","shell.execute_reply.started":"2021-07-24T18:27:18.377605Z","shell.execute_reply":"2021-07-24T18:27:18.386820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.388522Z","iopub.execute_input":"2021-07-24T18:27:18.388993Z","iopub.status.idle":"2021-07-24T18:27:18.400342Z","shell.execute_reply.started":"2021-07-24T18:27:18.388956Z","shell.execute_reply":"2021-07-24T18:27:18.399371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l  '../input/siims-c19-64x64-image-study-png/image/000a312787f2_image.png'","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:18.401372Z","iopub.execute_input":"2021-07-24T18:27:18.401684Z","iopub.status.idle":"2021-07-24T18:27:19.077763Z","shell.execute_reply.started":"2021-07-24T18:27:18.401620Z","shell.execute_reply":"2021-07-24T18:27:19.076979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"file_path\"] = df.apply(lambda x: f'../input/siims-c19-64x64-image-study-png/image/{x[\"id\"]}.png', axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.079206Z","iopub.execute_input":"2021-07-24T18:27:19.079816Z","iopub.status.idle":"2021-07-24T18:27:19.154354Z","shell.execute_reply.started":"2021-07-24T18:27:19.079771Z","shell.execute_reply":"2021-07-24T18:27:19.153706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.155252Z","iopub.execute_input":"2021-07-24T18:27:19.155608Z","iopub.status.idle":"2021-07-24T18:27:19.164997Z","shell.execute_reply.started":"2021-07-24T18:27:19.155572Z","shell.execute_reply":"2021-07-24T18:27:19.164191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into training and testing sets\ntrain_df, test_df, train_y, test_y = train_test_split(df,\n                                                   df['study_class'],\n                                                   stratify     = df['study_class'],\n                                                   test_size    = 0.33,\n                                                   random_state = 451)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.165926Z","iopub.execute_input":"2021-07-24T18:27:19.166143Z","iopub.status.idle":"2021-07-24T18:27:19.182256Z","shell.execute_reply.started":"2021-07-24T18:27:19.166123Z","shell.execute_reply":"2021-07-24T18:27:19.181420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.183469Z","iopub.execute_input":"2021-07-24T18:27:19.183870Z","iopub.status.idle":"2021-07-24T18:27:19.193224Z","shell.execute_reply.started":"2021-07-24T18:27:19.183833Z","shell.execute_reply":"2021-07-24T18:27:19.192414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Split once more, so that we may produce a validation set\n# #labels = train_df.pop('target')\n# train_df, valid_df, train_y, Valid_y = train_test_split(train_df,\n#                                                         train_df[\"study_class\"],\n#                                                         stratify     = train_df[\"study_class\"],\n#                                                         test_size    = 0.2,\n#                                                         random_state = 451)\n\n# # Reassemble labels\n# # train_df['target'] = train_y\n# # probe_df['target'] = probe_y","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.194255Z","iopub.execute_input":"2021-07-24T18:27:19.194497Z","iopub.status.idle":"2021-07-24T18:27:19.200369Z","shell.execute_reply.started":"2021-07-24T18:27:19.194468Z","shell.execute_reply":"2021-07-24T18:27:19.199635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_multiple_images(image_dataframe, rows = 4, columns = 4, figsize = (16, 20), preprocessing=None):\n    '''\n    Plots Multiple Images\n    Reads, resizes, applies preprocessing if desired and plots multiple images from a given dataframe\n    '''\n    image_dataframe = image_dataframe.reset_index(drop=True)\n    fig = plt.figure(figsize=figsize)\n    ax  = []\n\n    for i in range(rows * columns):\n        img = plt.imread(image_dataframe.loc[i,'file_path'])\n        #img = cv2.resize(img, resize)\n        \n        if preprocessing:\n            img = preprocessing(img)\n        \n        ax.append(fig.add_subplot(rows, columns, i+1) )\n        ax[-1].set_title(\"Xray \"+str(i+1))\n        plt.imshow(img, alpha=1, cmap='gray')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.201882Z","iopub.execute_input":"2021-07-24T18:27:19.202294Z","iopub.status.idle":"2021-07-24T18:27:19.212116Z","shell.execute_reply.started":"2021-07-24T18:27:19.202259Z","shell.execute_reply":"2021-07-24T18:27:19.211445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_multiple_images(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:19.213107Z","iopub.execute_input":"2021-07-24T18:27:19.213391Z","iopub.status.idle":"2021-07-24T18:27:20.970382Z","shell.execute_reply.started":"2021-07-24T18:27:19.213366Z","shell.execute_reply":"2021-07-24T18:27:20.966713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path, image_dims = (128,128), grayscale=True, flatten=True, interpolation = cv2.INTER_AREA):\n    '''\n    Loads an image, resizes and removes redudant channels if so desired\n    '''\n    image         = cv2.imread(image_path)\n    #resized_image = cv2.resize(image, image_dims, interpolation = interpolation)\n    resized_image = image\n    \n    if grayscale:\n        resized_image = resized_image[:,:,0]\n    \n    if flatten:\n        resized_image = resized_image.flatten()\n    \n    return(resized_image)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:20.971703Z","iopub.execute_input":"2021-07-24T18:27:20.972041Z","iopub.status.idle":"2021-07-24T18:27:20.977524Z","shell.execute_reply.started":"2021-07-24T18:27:20.972004Z","shell.execute_reply":"2021-07-24T18:27:20.976696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_flattened_dataframe(df, interpolation = cv2.INTER_AREA):\n    df     = df.reset_index(drop=True)\n    result = pd.DataFrame()\n    \n    for i in tqdm(range(df.shape[0])):\n        im_path = df.loc[i,'file_path']\n        current = load_image(im_path, interpolation = interpolation).tolist()\n        current = current\n        current = pd.DataFrame(current).T\n        result  = result.append(current)\n    \n    #result[\"study_class\"] = df[\"study_class\"]\n    \n    return(result)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:20.980486Z","iopub.execute_input":"2021-07-24T18:27:20.980859Z","iopub.status.idle":"2021-07-24T18:27:20.987184Z","shell.execute_reply.started":"2021-07-24T18:27:20.980830Z","shell.execute_reply":"2021-07-24T18:27:20.986472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flat_train_df = create_flattened_dataframe(train_df)\n#flat_valid_df = create_flattened_dataframe(valid_df)\nflat_test_df = create_flattened_dataframe(test_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:27:20.988573Z","iopub.execute_input":"2021-07-24T18:27:20.988821Z","iopub.status.idle":"2021-07-24T18:30:31.855373Z","shell.execute_reply.started":"2021-07-24T18:27:20.988798Z","shell.execute_reply":"2021-07-24T18:30:31.854565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flat_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:30:31.856897Z","iopub.execute_input":"2021-07-24T18:30:31.857246Z","iopub.status.idle":"2021-07-24T18:30:32.017961Z","shell.execute_reply.started":"2021-07-24T18:30:31.857209Z","shell.execute_reply":"2021-07-24T18:30:32.017010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parameters = {\n#     'cls__estimator__penalty': ['l2'],\n#     'cls__estimator__C': [1, 5, 7, 10],\n#     'cls__estimator__max_iter': [50, 100, 300, 500],\n#     'cls__estimator__solver' : ['lbfgs'],\n# }","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:30:32.018992Z","iopub.execute_input":"2021-07-24T18:30:32.019259Z","iopub.status.idle":"2021-07-24T18:30:32.022253Z","shell.execute_reply.started":"2021-07-24T18:30:32.019234Z","shell.execute_reply":"2021-07-24T18:30:32.021375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\n# # define model\n# lr = LogisticRegression(class_weight='balanced', n_jobs=32)\n# #classifier = OneVsRestClassifier(lr, n_jobs=32)\n# pipeline = Pipeline([\n#     ('cls', LogisticRegression(class_weight='balanced', n_jobs=32), n_jobs=32),\n# ])\n\ngrid = {\n    \"C\": [0.1, 1, 10, 100], \n    \"class_weight\": ['none'], \n    \"penalty\":[\"l1\",\"l2\",\"none\"],\n    \"solver\":[\"saga\"]\n}# l1 lasso l2 ridge\n\nlr = LogisticRegression(n_jobs=32)\nlr_cv = GridSearchCV(lr, grid, cv = 3, verbose=3)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:30:32.023723Z","iopub.execute_input":"2021-07-24T18:30:32.023992Z","iopub.status.idle":"2021-07-24T18:30:32.032956Z","shell.execute_reply.started":"2021-07-24T18:30:32.023969Z","shell.execute_reply":"2021-07-24T18:30:32.032181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%time\n# skf = StratifiedKFold(n_splits=2)\n#grid_search_tune = GridSearchCV(pipeline, parameters, cv=skf.split(flat_train_df, train_df[\"study_class\"]),  verbose=3)\nlr_cv.fit(flat_train_df, train_df[\"study_class\"])\n\nprint(\"Best Score: \", lr_cv.best_score_)\nprint(\"Best Params: \", lr_cv.best_params_)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:30:32.034014Z","iopub.execute_input":"2021-07-24T18:30:32.034253Z","iopub.status.idle":"2021-07-24T18:40:09.984278Z","shell.execute_reply.started":"2021-07-24T18:30:32.034233Z","shell.execute_reply":"2021-07-24T18:40:09.983258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create Logistic Regression\n# #logit_model = LogisticRegression(random_state=451, solver='lbfgs', n_jobs=-1)\n\n# #from sklearn.multiclass import OneVsRestClassifier\n# #from sklearn.linear_model import LogisticRegressionCV\n\n# lr = LogisticRegression(class_weight='none',\n#                         C=1,\n#                         penalty='l2',\n#                         max_iter=500,\n#                         solver='saga',\n#                         n_jobs=32)\n# lr.fit(flat_train_df, train_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:46:58.277425Z","iopub.execute_input":"2021-07-24T18:46:58.277798Z","iopub.status.idle":"2021-07-24T18:46:58.281265Z","shell.execute_reply.started":"2021-07-24T18:46:58.277769Z","shell.execute_reply":"2021-07-24T18:46:58.280499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# logit_preds_val  = lr.predict_proba(flat_test_df)\n# evaluate_predictions(logit_preds_val[:,1], eval_df = test_df)\n# lr.score(flat_test_df, test_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:46:59.647704Z","iopub.execute_input":"2021-07-24T18:46:59.648031Z","iopub.status.idle":"2021-07-24T18:46:59.651056Z","shell.execute_reply.started":"2021-07-24T18:46:59.648004Z","shell.execute_reply":"2021-07-24T18:46:59.650147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:40:10.101575Z","iopub.status.idle":"2021-07-24T18:40:10.102054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_predictions(preds, eval_df = test_df):\n    '''\n    Evaluate Predictions Function\n    Returns accuracy and auc of the model\n    '''\n    auroc = roc_auc_score(eval_df['study_class'].astype('uint8'), preds)\n    accur = accuracy_score(eval_df['study_class'].astype('uint8'), preds >= 0.5)\n    print('Accuracy: ' + str(auroc))\n    print('AUC: ' + str(accur))","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:41:19.603133Z","iopub.execute_input":"2021-07-24T18:41:19.603453Z","iopub.status.idle":"2021-07-24T18:41:19.608361Z","shell.execute_reply.started":"2021-07-24T18:41:19.603422Z","shell.execute_reply":"2021-07-24T18:41:19.607339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Evaluate Model Results - Validation Set\n# logit_preds_val  = logit_model.predict_proba(flat_valid_df)\n# #evaluate_predictions(logit_preds_val[:,1], eval_df = valid_df)\n# logit_model.score(flat_valid_df, valid_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:40:10.104647Z","iopub.status.idle":"2021-07-24T18:40:10.105075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate Model Results - Validation Set\nlogit_preds_val  = lr_cv.predict_proba(flat_test_df)\nevaluate_predictions(logit_preds_val[:,1], eval_df = test_df)\nlr_cv.score(flat_test_df, test_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:41:55.889958Z","iopub.execute_input":"2021-07-24T18:41:55.890279Z","iopub.status.idle":"2021-07-24T18:41:56.130465Z","shell.execute_reply.started":"2021-07-24T18:41:55.890250Z","shell.execute_reply":"2021-07-24T18:41:56.128849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\n\n#Creating matplotlib axes object to assign figuresize and figure title\nfig, ax = plt.subplots(figsize=(10, 6))\nax.set_title('Confusion Matrx')\n\ndisp = metrics.plot_confusion_matrix(lr_cv, flat_test_df, test_df[\"study_class\"], ax = ax)\ndisp.confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2021-07-24T18:42:04.237435Z","iopub.execute_input":"2021-07-24T18:42:04.237788Z","iopub.status.idle":"2021-07-24T18:42:04.568291Z","shell.execute_reply.started":"2021-07-24T18:42:04.237756Z","shell.execute_reply":"2021-07-24T18:42:04.567358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}