{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport cv2\n\npd.set_option('display.max_columns', None)  \npd.set_option('display.max_colwidth', None)\n\nimport matplotlib.pyplot as plt\nimport matplotlib.style  as style\n\nfrom tqdm  import tqdm\nfrom sklearn.metrics     import accuracy_score, roc_auc_score\nfrom sklearn.linear_model      import LogisticRegression\nfrom sklearn.model_selection   import train_test_split","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-25T04:02:46.563478Z","iopub.execute_input":"2021-07-25T04:02:46.564070Z","iopub.status.idle":"2021-07-25T04:02:46.820717Z","shell.execute_reply.started":"2021-07-25T04:02:46.564033Z","shell.execute_reply":"2021-07-25T04:02:46.819838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install scikit-learn  -U","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:02:46.822421Z","iopub.execute_input":"2021-07-25T04:02:46.822843Z","iopub.status.idle":"2021-07-25T04:03:02.252169Z","shell.execute_reply.started":"2021-07-25T04:02:46.822800Z","shell.execute_reply":"2021-07-25T04:03:02.250969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.255050Z","iopub.execute_input":"2021-07-25T04:03:02.255496Z","iopub.status.idle":"2021-07-25T04:03:02.283024Z","shell.execute_reply.started":"2021-07-25T04:03:02.255445Z","shell.execute_reply":"2021-07-25T04:03:02.282012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.284807Z","iopub.execute_input":"2021-07-25T04:03:02.285201Z","iopub.status.idle":"2021-07-25T04:03:02.293341Z","shell.execute_reply.started":"2021-07-25T04:03:02.285160Z","shell.execute_reply":"2021-07-25T04:03:02.292583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.294553Z","iopub.execute_input":"2021-07-25T04:03:02.294802Z","iopub.status.idle":"2021-07-25T04:03:02.328221Z","shell.execute_reply.started":"2021-07-25T04:03:02.294778Z","shell.execute_reply":"2021-07-25T04:03:02.327346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"] = df[label_cols].apply(lambda x: df.columns[x.argmax()+1], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.331274Z","iopub.execute_input":"2021-07-25T04:03:02.331549Z","iopub.status.idle":"2021-07-25T04:03:02.516712Z","shell.execute_reply.started":"2021-07-25T04:03:02.331523Z","shell.execute_reply":"2021-07-25T04:03:02.515657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.drop(df.columns[1:5], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.517971Z","iopub.execute_input":"2021-07-25T04:03:02.518262Z","iopub.status.idle":"2021-07-25T04:03:02.527548Z","shell.execute_reply.started":"2021-07-25T04:03:02.518233Z","shell.execute_reply":"2021-07-25T04:03:02.526406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.530206Z","iopub.execute_input":"2021-07-25T04:03:02.530507Z","iopub.status.idle":"2021-07-25T04:03:02.545039Z","shell.execute_reply.started":"2021-07-25T04:03:02.530475Z","shell.execute_reply":"2021-07-25T04:03:02.543900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"].unique()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.547918Z","iopub.execute_input":"2021-07-25T04:03:02.548369Z","iopub.status.idle":"2021-07-25T04:03:02.557406Z","shell.execute_reply.started":"2021-07-25T04:03:02.548324Z","shell.execute_reply":"2021-07-25T04:03:02.556173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_class_to_num = {\"Typical Appearance\":0, \"Atypical Appearance\":1, \"Negative for Pneumonia\":2, \"Indeterminate Appearance\":3}","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.558436Z","iopub.execute_input":"2021-07-25T04:03:02.558682Z","iopub.status.idle":"2021-07-25T04:03:02.567341Z","shell.execute_reply.started":"2021-07-25T04:03:02.558658Z","shell.execute_reply":"2021-07-25T04:03:02.566120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study_class_to_num","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.569042Z","iopub.execute_input":"2021-07-25T04:03:02.569362Z","iopub.status.idle":"2021-07-25T04:03:02.580370Z","shell.execute_reply.started":"2021-07-25T04:03:02.569333Z","shell.execute_reply":"2021-07-25T04:03:02.579441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"study_class\"] = df[\"study_class\"].apply(lambda x: study_class_to_num[x])","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.581647Z","iopub.execute_input":"2021-07-25T04:03:02.582003Z","iopub.status.idle":"2021-07-25T04:03:02.594092Z","shell.execute_reply.started":"2021-07-25T04:03:02.581969Z","shell.execute_reply":"2021-07-25T04:03:02.592786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.595421Z","iopub.execute_input":"2021-07-25T04:03:02.595805Z","iopub.status.idle":"2021-07-25T04:03:02.611158Z","shell.execute_reply.started":"2021-07-25T04:03:02.595776Z","shell.execute_reply":"2021-07-25T04:03:02.610197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -l  '../input/siims-c19-64x64-image-study-png/study/00086460a852_study.png'","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:02.612244Z","iopub.execute_input":"2021-07-25T04:03:02.612658Z","iopub.status.idle":"2021-07-25T04:03:03.442415Z","shell.execute_reply.started":"2021-07-25T04:03:02.612619Z","shell.execute_reply":"2021-07-25T04:03:03.441301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"file_path\"] = df.apply(lambda x: f'../input/siims-c19-64x64-image-study-png/study/{x[\"id\"]}.png', axis=1)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.444330Z","iopub.execute_input":"2021-07-25T04:03:03.444769Z","iopub.status.idle":"2021-07-25T04:03:03.501932Z","shell.execute_reply.started":"2021-07-25T04:03:03.444720Z","shell.execute_reply":"2021-07-25T04:03:03.501002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.503341Z","iopub.execute_input":"2021-07-25T04:03:03.503697Z","iopub.status.idle":"2021-07-25T04:03:03.513790Z","shell.execute_reply.started":"2021-07-25T04:03:03.503660Z","shell.execute_reply":"2021-07-25T04:03:03.513093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into training and testing sets\ntrain_df, test_df, train_y, test_y = train_test_split(df,\n                                                   df['study_class'],\n                                                   stratify     = df['study_class'],\n                                                   test_size    = 0.2,\n                                                   random_state = 451)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.514910Z","iopub.execute_input":"2021-07-25T04:03:03.515343Z","iopub.status.idle":"2021-07-25T04:03:03.534161Z","shell.execute_reply.started":"2021-07-25T04:03:03.515314Z","shell.execute_reply":"2021-07-25T04:03:03.532987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.535433Z","iopub.execute_input":"2021-07-25T04:03:03.535697Z","iopub.status.idle":"2021-07-25T04:03:03.546519Z","shell.execute_reply.started":"2021-07-25T04:03:03.535671Z","shell.execute_reply":"2021-07-25T04:03:03.545284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Split once more, so that we may produce a validation set\n# #labels = train_df.pop('target')\n# train_df, valid_df, train_y, Valid_y = train_test_split(train_df,\n#                                                         train_df[\"study_class\"],\n#                                                         stratify     = train_df[\"study_class\"],\n#                                                         test_size    = 0.2,\n#                                                         random_state = 451)\n\n# # Reassemble labels\n# # train_df['target'] = train_y\n# # probe_df['target'] = probe_y","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.547735Z","iopub.execute_input":"2021-07-25T04:03:03.548221Z","iopub.status.idle":"2021-07-25T04:03:03.552004Z","shell.execute_reply.started":"2021-07-25T04:03:03.548184Z","shell.execute_reply":"2021-07-25T04:03:03.550969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_multiple_images(image_dataframe, rows = 4, columns = 4, figsize = (16, 20), preprocessing=None):\n    '''\n    Plots Multiple Images\n    Reads, resizes, applies preprocessing if desired and plots multiple images from a given dataframe\n    '''\n    image_dataframe = image_dataframe.reset_index(drop=True)\n    fig = plt.figure(figsize=figsize)\n    ax  = []\n\n    for i in range(rows * columns):\n        img = plt.imread(image_dataframe.loc[i,'file_path'])\n        #img = cv2.resize(img, resize)\n        \n        if preprocessing:\n            img = preprocessing(img)\n        \n        ax.append(fig.add_subplot(rows, columns, i+1) )\n        ax[-1].set_title(\"Xray \"+str(i+1))\n        plt.imshow(img, alpha=1, cmap='gray')\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.553585Z","iopub.execute_input":"2021-07-25T04:03:03.554024Z","iopub.status.idle":"2021-07-25T04:03:03.565108Z","shell.execute_reply.started":"2021-07-25T04:03:03.553974Z","shell.execute_reply":"2021-07-25T04:03:03.564136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_multiple_images(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:03.566272Z","iopub.execute_input":"2021-07-25T04:03:03.566532Z","iopub.status.idle":"2021-07-25T04:03:05.517329Z","shell.execute_reply.started":"2021-07-25T04:03:03.566507Z","shell.execute_reply":"2021-07-25T04:03:05.516282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_image(image_path, image_dims = (128,128), grayscale=True, flatten=True, interpolation = cv2.INTER_AREA):\n    '''\n    Loads an image, resizes and removes redudant channels if so desired\n    '''\n    image         = cv2.imread(image_path)\n    #resized_image = cv2.resize(image, image_dims, interpolation = interpolation)\n    resized_image = image\n    \n    if grayscale:\n        resized_image = resized_image[:,:,0]\n    \n    if flatten:\n        resized_image = resized_image.flatten()\n    \n    return(resized_image)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:05.518604Z","iopub.execute_input":"2021-07-25T04:03:05.518921Z","iopub.status.idle":"2021-07-25T04:03:05.524285Z","shell.execute_reply.started":"2021-07-25T04:03:05.518887Z","shell.execute_reply":"2021-07-25T04:03:05.523169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_flattened_dataframe(df, interpolation = cv2.INTER_AREA):\n    df     = df.reset_index(drop=True)\n    result = pd.DataFrame()\n    \n    for i in tqdm(range(df.shape[0])):\n        im_path = df.loc[i,'file_path']\n        current = load_image(im_path, interpolation = interpolation).tolist()\n        current = current\n        current = pd.DataFrame(current).T\n        result  = result.append(current)\n    \n    #result[\"study_class\"] = df[\"study_class\"]\n    \n    return(result)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:05.528625Z","iopub.execute_input":"2021-07-25T04:03:05.529046Z","iopub.status.idle":"2021-07-25T04:03:05.537822Z","shell.execute_reply.started":"2021-07-25T04:03:05.529011Z","shell.execute_reply":"2021-07-25T04:03:05.536654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flat_train_df = create_flattened_dataframe(train_df)\n#flat_valid_df = create_flattened_dataframe(valid_df)\nflat_test_df = create_flattened_dataframe(test_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:03:05.539877Z","iopub.execute_input":"2021-07-25T04:03:05.540298Z","iopub.status.idle":"2021-07-25T04:06:30.112024Z","shell.execute_reply.started":"2021-07-25T04:03:05.540252Z","shell.execute_reply":"2021-07-25T04:06:30.111010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flat_train_df.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.113485Z","iopub.execute_input":"2021-07-25T04:06:30.113947Z","iopub.status.idle":"2021-07-25T04:06:30.403960Z","shell.execute_reply.started":"2021-07-25T04:06:30.113911Z","shell.execute_reply":"2021-07-25T04:06:30.402865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# parameters = {\n#     'cls__estimator__penalty': ['l1','l2', 'none'],\n#     'cls__estimator__C': [0.1, 1, 5],\n#     'cls__estimator__max_iter': [50, 100, 300],\n#     'cls__estimator__solver' : ['lbfgs','saga','liblinear'],\n# }","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.405366Z","iopub.execute_input":"2021-07-25T04:06:30.405741Z","iopub.status.idle":"2021-07-25T04:06:30.409862Z","shell.execute_reply.started":"2021-07-25T04:06:30.405700Z","shell.execute_reply":"2021-07-25T04:06:30.408999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import GridSearchCV, StratifiedShuffleSplit, StratifiedKFold\n# define model\n#lr = LogisticRegression(class_weight='balanced', n_jobs=32)\n#classifier = OneVsRestClassifier(lr, n_jobs=32)\n#pipeline = Pipeline([\n#    ('cls', OneVsRestClassifier(LogisticRegression(class_weight='balanced', n_jobs=32), n_jobs=32)),\n#])\n\n# grid = {\n#     \"C\": [0.1, 1], \n#     \"class_weight\": ['None','balanced'], \n#     \"penalty\":[\"l1\",\"l2\",\"none\"],\n#     \"solver\":['liblinear','lbfgs']\n# }# l1 lasso l2 ridge\n\ngrid = {\n    \"C\": [0.1, 1], \n    \"penalty\":[\"l1\"],\n    \"solver\":[\"saga\"]\n}# l1 lasso l2 ridge\n\n\nlr = LogisticRegression(class_weight='balanced',n_jobs=32)\nskf = StratifiedKFold(n_splits=5)\nlogit_model = GridSearchCV(lr, grid, cv=skf.split(flat_train_df, train_df['study_class']), verbose=3)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.411225Z","iopub.execute_input":"2021-07-25T04:06:30.411518Z","iopub.status.idle":"2021-07-25T04:06:30.429990Z","shell.execute_reply.started":"2021-07-25T04:06:30.411488Z","shell.execute_reply":"2021-07-25T04:06:30.428905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create Logistic Regression\n# # logit_model = LogisticRegression(random_state=451, multi_class='ovr', solver='liblinear')\n# from sklearn.multiclass import OneVsRestClassifier\n# from sklearn.linear_model import LogisticRegressionCV\n\n# lr = LogisticRegression(class_weight='none',\n#                         C=5,\n#                         penalty='l2',\n#                         max_iter=500,\n#                         solver='lbfgs',\n#                         n_jobs=32)\n# logit_model = OneVsRestClassifier(lr, n_jobs=32)","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.431432Z","iopub.execute_input":"2021-07-25T04:06:30.431914Z","iopub.status.idle":"2021-07-25T04:06:30.441451Z","shell.execute_reply.started":"2021-07-25T04:06:30.431874Z","shell.execute_reply":"2021-07-25T04:06:30.440445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# logit_model","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.442636Z","iopub.execute_input":"2021-07-25T04:06:30.443001Z","iopub.status.idle":"2021-07-25T04:06:30.453150Z","shell.execute_reply.started":"2021-07-25T04:06:30.442972Z","shell.execute_reply":"2021-07-25T04:06:30.452219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logit_model.fit(flat_train_df, train_df['study_class'])","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:06:30.454334Z","iopub.execute_input":"2021-07-25T04:06:30.454591Z","iopub.status.idle":"2021-07-25T04:24:23.627596Z","shell.execute_reply.started":"2021-07-25T04:06:30.454565Z","shell.execute_reply":"2021-07-25T04:24:23.626277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_predictions(preds, eval_df = test_df):\n    '''\n    Evaluate Predictions Function\n    Returns accuracy and auc of the model\n    '''\n    auroc = roc_auc_score(eval_df['study_class'].astype('uint8'), preds)\n    accur = accuracy_score(eval_df['study_class'].astype('uint8'), preds >= 0.5)\n    print('Accuracy: ' + str(auroc))\n    print('AUC: ' + str(accur))","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:24:23.629897Z","iopub.execute_input":"2021-07-25T04:24:23.630325Z","iopub.status.idle":"2021-07-25T04:24:23.636185Z","shell.execute_reply.started":"2021-07-25T04:24:23.630277Z","shell.execute_reply":"2021-07-25T04:24:23.635256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Evaluate Model Results - Validation Set\n# logit_preds_val  = logit_model.predict_proba(flat_valid_df)\n# #evaluate_predictions(logit_preds_val[:,1], eval_df = valid_df)\n# logit_model.score(flat_valid_df, valid_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:24:23.637462Z","iopub.execute_input":"2021-07-25T04:24:23.637946Z","iopub.status.idle":"2021-07-25T04:24:23.655356Z","shell.execute_reply.started":"2021-07-25T04:24:23.637912Z","shell.execute_reply":"2021-07-25T04:24:23.654084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate Model Results - Validation Set\nlogit_preds_val  = logit_model.predict_proba(flat_test_df)\n#evaluate_predictions(logit_preds_val[:,1], eval_df = valid_df)\nlogit_model.score(flat_test_df, test_df[\"study_class\"])","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:24:23.656703Z","iopub.execute_input":"2021-07-25T04:24:23.657090Z","iopub.status.idle":"2021-07-25T04:24:23.831702Z","shell.execute_reply.started":"2021-07-25T04:24:23.657046Z","shell.execute_reply":"2021-07-25T04:24:23.830472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import metrics\n\n#Creating matplotlib axes object to assign figuresize and figure title\nfig, ax = plt.subplots(figsize=(10, 6))\nax.set_title('Confusion Matrx')\n\ndisp = metrics.plot_confusion_matrix(logit_model, flat_test_df, test_df[\"study_class\"], ax = ax)\ndisp.confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2021-07-25T04:24:23.833661Z","iopub.execute_input":"2021-07-25T04:24:23.834171Z","iopub.status.idle":"2021-07-25T04:24:24.227695Z","shell.execute_reply.started":"2021-07-25T04:24:23.834120Z","shell.execute_reply":"2021-07-25T04:24:24.226947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}