{"cells":[{"metadata":{"_uuid":"f962cd52a6e0ab988871cdd98a53642124fd9999"},"cell_type":"markdown","source":"# Practical EDA on numerical data"},{"metadata":{"trusted":true,"_uuid":"a8ff7cf5fc7b75312eaf75e540380b7cbfeef5ed"},"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"889b813cf8efc95a3dabf73dbf521a90428dc62c"},"cell_type":"code","source":"import pydicom","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a0488b9ee174f0d170f8944c9d9cfab23070946"},"cell_type":"code","source":"import gc\nimport warnings\nwarnings.simplefilter(action = 'ignore')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6c2af121fc76c5ac2ff3b123f58b34d443fbd51"},"cell_type":"code","source":"from lightgbm import LGBMRegressor, LGBMClassifier\nfrom sklearn.metrics import roc_auc_score, mean_absolute_error\nfrom sklearn.model_selection import KFold","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a77183817d409f9c21c3ec0f1690faee08cfcc1e"},"cell_type":"markdown","source":"## Analisys of base information"},{"metadata":{"_uuid":"fde5175d287c72835ed0778f3b52970cb0d8a9a9"},"cell_type":"markdown","source":"### Loading data"},{"metadata":{"_uuid":"886f9e9116b7fe7f1b22d5bd3569857223a02b36","trusted":true},"cell_type":"code","source":"detailed_class_info = pd.read_csv('../input/stage_1_detailed_class_info.csv')\ntrain_labels = pd.read_csv('../input/stage_1_train_labels.csv')\n\ndf = pd.merge(left = detailed_class_info, right = train_labels, how = 'left', on = 'patientId')\n\ndel detailed_class_info, train_labels\ngc.collect()\n\ndf.info(null_counts = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0edbfa081005b9479d76ad450950eb9e178ae426"},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d883039f1bf01f56531015be098fe7e0782ee274"},"cell_type":"code","source":"df = df.drop_duplicates()\ndf.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"27e599082f02d9a070580017b844900d48e84012"},"cell_type":"markdown","source":"### Rows per patientID"},{"metadata":{"trusted":true,"_uuid":"987031be78f4a0ac70a7bf9598326983e98e0110"},"cell_type":"code","source":"df['patientId'].value_counts().head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cfaee1b30372a329203b99e3f74e01e1492a0869"},"cell_type":"code","source":"df[df['patientId'] == '32408669-c137-4e8d-bd62-fe8345b40e73']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"217746fb190a19b24908ca9ae2fe01ead8af5255"},"cell_type":"markdown","source":"Count of rows per patientID has 4 values:"},{"metadata":{"trusted":true,"_uuid":"188d6bccfacf201b9e628d4ec9bac633630022a4"},"cell_type":"code","source":"df['patientId'].value_counts().value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ba70e77bc8ee04cc4c97ea983b8ed7f4647734d"},"cell_type":"markdown","source":"Each of patients without pneumonia has only one row in dataset:"},{"metadata":{"trusted":true,"_uuid":"5081ac21124d4c7faf0a6d72d02cb0d7d3a37d7c"},"cell_type":"code","source":"df[df['Target'] == 0]['patientId'].value_counts().value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b0e1d3c19ad2ad2193e7315b38f461af15363877"},"cell_type":"markdown","source":"### Distribution of `class`"},{"metadata":{"trusted":true,"_uuid":"a95bf2b15015a8c9dd50da03286063e7c6174023"},"cell_type":"code","source":"sns.countplot(x = 'class', hue = 'Target', data = df);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3ee022cc1ffbd3df109b5129f7c74049c70d1000"},"cell_type":"code","source":"df[df['class'] == 'Lung Opacity']['Target'].value_counts(dropna = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"57bec2ad7a9b28b9c331d84007aa9937948934ec"},"cell_type":"code","source":"df[df['class'] == 'No Lung Opacity / Not Normal']['Target'].value_counts(dropna = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46eb644b315439a48a601b06a7013439e094e09d"},"cell_type":"code","source":"df[df['class'] == 'Normal']['Target'].value_counts(dropna = False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"89f1a7b92aa08c1d9eed309bac306d361da25698"},"cell_type":"markdown","source":"Only class `Lung Opacity` has pneumonia on the train set."},{"metadata":{"trusted":true,"_uuid":"b96c89da7131fcd72a209d8e599e59b3e608a06a"},"cell_type":"code","source":"print('Patients can have {} different classes'.format(df.groupby('patientId')['class'].nunique().nunique()))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8a0f3dccf509e5950d9da8840d79facd3be53992"},"cell_type":"markdown","source":"That is, \"class == Lung Opacity\" is equivalent to \"Target == 1\" or \"image has pneumonia areas\"."},{"metadata":{"_uuid":"29ddeb2cba9038b9564fd2093b62bc924acfc5c9"},"cell_type":"markdown","source":"### Spatial features: x, y, width, height"},{"metadata":{"trusted":true,"_uuid":"50efce97ab973305cedc01f0c8636a15971c9440"},"cell_type":"code","source":"df_areas = df.dropna()[['x', 'y', 'width', 'height']].copy()\ndf_areas['x_2'] = df_areas['x'] + df_areas['width']\ndf_areas['y_2'] = df_areas['y'] + df_areas['height']\ndf_areas['x_center'] = df_areas['x'] + df_areas['width'] / 2\ndf_areas['y_center'] = df_areas['y'] + df_areas['height'] / 2\ndf_areas['area'] = df_areas['width'] * df_areas['height']\n\ndf_areas.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5250b7c6a0f59946f8d886b1115f8b0b2a551dac"},"cell_type":"code","source":"sns.jointplot(x = 'x', y = 'y', data = df_areas, kind = 'hex', gridsize = 20);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b91c449b82eb6a79a4c5523d0e1e568267ca5a4"},"cell_type":"code","source":"sns.jointplot(x = 'x_center', y = 'y_center', data = df_areas, kind = 'hex', gridsize = 20);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"049001065b69d1e156634b48c5414079d10738ba"},"cell_type":"code","source":"sns.jointplot(x = 'x_2', y = 'y_2', data = df_areas, kind = 'hex', gridsize = 20);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c00bcb47f427b1496c04323bf826633acd832f94"},"cell_type":"markdown","source":"Centers and opposite corners have density more (variance less), than main corners (x, y). The centers have a high density and small correlation. \n\nThere is no reason to replace (x, y) with (x_center, y_center) or (x_2, y_2)."},{"metadata":{"trusted":true,"_uuid":"a80f2921361988b5062dd6e6c9390222a1b3477c"},"cell_type":"code","source":"sns.jointplot(x = 'width', y = 'height', data = df_areas, kind = 'hex', gridsize = 20);","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1e9e87de376d83d588627a0b16adecdb49ee570"},"cell_type":"markdown","source":"Widths and heights have a very hight correlation."},{"metadata":{"trusted":true,"_uuid":"298381ab3af8d4cb1cc851f28fa8df372cc3c3db"},"cell_type":"code","source":"n_columns = 3\nn_rows = 3\n_, axes = plt.subplots(n_rows, n_columns, figsize=(8 * n_columns, 5 * n_rows))\nfor i, c in enumerate(df_areas.columns):\n    sns.boxplot(y = c, data = df_areas, ax = axes[i // n_columns, i % n_columns])\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4571ebe610cd4e2c37042ea0e0136ef0cb517e6f"},"cell_type":"markdown","source":"There are some outliers, especially for 'width' and 'height' features."},{"metadata":{"trusted":true,"_uuid":"406b4ad6353479c4e22529a38ee9c1d9122a1b70"},"cell_type":"code","source":"df_areas[df_areas['width'] > 500]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1031bf69bd8e194cc1ccfb4a7867ef7394ca16a3"},"cell_type":"code","source":"pid_width = list(df[df['width'] > 500]['patientId'].values)\ndf[df['patientId'].isin(pid_width)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"9fef1115524575057d9dca36246235ae65f69e55"},"cell_type":"markdown","source":"One patient. Row can be dropped."},{"metadata":{"trusted":true,"_uuid":"9ad6f04844ec70b0755610d57270901f74bbce04"},"cell_type":"code","source":"df_areas[df_areas['height'] > 900].shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c798ccb01d685dad7db13902cc0786e2e310cb1"},"cell_type":"code","source":"pid_height = list(df[df['height'] > 900]['patientId'].values)\ndf[df['patientId'].isin(pid_height)]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2bb8c6e500c2f28539d4c5668367f2ff6ff8d7b0"},"cell_type":"markdown","source":"Two patients. All rows must be dropped together."},{"metadata":{"trusted":true,"_uuid":"5a96753bf308736ba52cbaabf6ef77c0620bdcd9"},"cell_type":"code","source":"df = df[~df['patientId'].isin(pid_width + pid_height)]\ndf.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb9d74f5063b7680e0d9abaea9333744d2e2dc3f"},"cell_type":"markdown","source":"## Analisys of meta information"},{"metadata":{"trusted":true,"_uuid":"57bda165ca1e8a7042ec9afafe8e6597fafbe219"},"cell_type":"code","source":"df_meta = df.drop('class', axis = 1).copy()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"857d97c138fbdc76b4a4b07f5eb06ce31e05d366"},"cell_type":"code","source":"dcm_columns = None\n\nfor n, pid in enumerate(df_meta['patientId'].unique()):\n    dcm_file = '../input/stage_1_train_images/%s.dcm' % pid\n    dcm_data = pydicom.read_file(dcm_file)\n    \n    if not dcm_columns:\n        dcm_columns = dcm_data.dir()\n        dcm_columns.remove('PixelSpacing')\n        dcm_columns.remove('PixelData')\n    \n    for col in dcm_columns:\n        if not (col in df_meta.columns):\n            df_meta[col] = np.nan\n        index = df_meta[df_meta['patientId'] == pid].index\n        df_meta.loc[index, col] = dcm_data.data_element(col).value\n        \n    del dcm_data\n    \ngc.collect()\n\ndf_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"593c70c46d2942917a5b496ceae46d4fe1772ce5"},"cell_type":"code","source":"to_drop = df_meta.nunique()\nto_drop = to_drop[(to_drop <= 1) | (to_drop == to_drop['patientId'])].index\nto_drop = to_drop.drop('patientId')\nto_drop","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60e90784acbfd6d7fcef44165cc6a9bafd5f6f07"},"cell_type":"code","source":"df_meta.drop(to_drop, axis = 1, inplace = True)\ndf_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a5c4737e9956abf99c4f36fb979dbe16e1e87cd"},"cell_type":"code","source":"print('Dropped {} useless features'.format(len(to_drop)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aa96e3d31c0fe03ff64779be9ed62a4b264bcbb1"},"cell_type":"code","source":"df_meta.nunique()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"edd2a0ec6e9e858655abdd3d738b894b813c35f1"},"cell_type":"code","source":"sum(df_meta['ReferringPhysicianName'].unique() != '')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d6f8724545d9f5cf2126afd0ae26bfb6ebe2af49"},"cell_type":"code","source":"df_meta.drop('ReferringPhysicianName', axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"16741bc10b086529d088756a03d723d6a6007760"},"cell_type":"markdown","source":"Dropped one more useless feature"},{"metadata":{"trusted":true,"_uuid":"c5dee5962e59074f3624f87a0294b2a26ac204d2"},"cell_type":"code","source":"df_meta['PatientAge'] = df_meta['PatientAge'].astype(int)\ndf_meta['SeriesDescription'] = df_meta['SeriesDescription'].map({'view: AP': 'AP', 'view: PA': 'PA'})\ndf_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8d7be023debb99957972b802d1986e6620d3508"},"cell_type":"code","source":"print('There are {} equal elements between SeriesDescription and ViewPosition from {}.' \\\n      .format(sum(df_meta['SeriesDescription'] == df_meta['ViewPosition']), df_meta.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e57a2ca4996badb6a388996f77b69f578ed34b0"},"cell_type":"code","source":"df_meta.drop('SeriesDescription', axis = 1, inplace = True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a520ccda233b35f6455ff076d6f67150fe16f926"},"cell_type":"markdown","source":"Dropped one feature wich is equal another."},{"metadata":{"trusted":true,"_uuid":"564b8c9f88175c48e084e38c6083d2db9add13f6"},"cell_type":"code","source":"plt.figure(figsize = (25, 5))\nsns.countplot(x = 'PatientAge', hue = 'Target', data = df_meta);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb18e22699f05a8f20d1a4aaeeb5a3ea5114b8ef"},"cell_type":"code","source":"sns.countplot(x = 'PatientSex', hue = 'Target', data = df_meta);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b5811d7f869199f1906a63393412f1922b3b768a"},"cell_type":"code","source":"sns.countplot(x = 'ViewPosition', hue = 'Target', data = df_meta);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6af2ec311df33c326bd9273bc7265055cb1e29b0"},"cell_type":"code","source":"df_meta['PatientSex'] = df_meta['PatientSex'].map({'F': 0, 'M': 1})\ndf_meta['ViewPosition'] = df_meta['ViewPosition'].map({'PA': 0, 'AP': 1})\ndf_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c59e937b46d0ffc1ce8cb45f5ca0bcb0980e957"},"cell_type":"code","source":"df_meta.corr()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1e3678e6d9117110306fb245dad422551f43174b"},"cell_type":"markdown","source":"'ViewPosition' have a high correlation with 'Target' and 'height' features. It can be useful..."},{"metadata":{"_uuid":"0b3ec798a886860ebc82f432588dd070e61597f1"},"cell_type":"markdown","source":"## Attempt of forecasting of target and spatial variables according to meta information"},{"metadata":{"trusted":true,"_uuid":"f432f86cd636f23158795f5435ae9ef02f5ea0ff"},"cell_type":"code","source":"def fast_lgbm_cv_scores(df, target, task, rs = 0):\n    warnings.simplefilter('ignore')\n    \n    if task == 'classification':\n        clf = LGBMClassifier(n_estimators = 10000, nthread = 4, random_state = rs)\n        metric = 'auc'\n    else:\n        clf = LGBMRegressor(n_estimators = 10000, nthread = 4, random_state = rs)\n        metric = 'mean_absolute_error'\n\n    # Cross validation model\n    folds = KFold(n_splits = 2, shuffle = True, random_state = rs)\n        \n    # Create arrays and dataframes to store results\n    pred = np.zeros(df.shape[0])\n    \n    feats = df.columns.drop(target)\n    \n    feature_importance_df = pd.DataFrame(index = feats)\n    \n    for n_fold, (train_idx, valid_idx) in enumerate(folds.split(df[feats], df[target])):\n        train_x, train_y = df[feats].iloc[train_idx], df[target].iloc[train_idx]\n        valid_x, valid_y = df[feats].iloc[valid_idx], df[target].iloc[valid_idx]\n\n        clf.fit(train_x, train_y, \n                eval_set = [(valid_x, valid_y)], eval_metric = metric, \n                verbose = -1, early_stopping_rounds = 100)\n\n        if task == 'classification':\n            pred[valid_idx] = clf.predict_proba(valid_x, num_iteration = clf.best_iteration_)[:, 1]\n        else:\n            pred[valid_idx] = clf.predict(valid_x, num_iteration = clf.best_iteration_)\n        \n        feature_importance_df[n_fold] = pd.Series(clf.feature_importances_, index = feats)\n        \n        del train_x, train_y, valid_x, valid_y\n        gc.collect()\n\n    if task == 'classification':    \n        return feature_importance_df, pred, roc_auc_score(df[target], pred)\n    else:\n        return feature_importance_df, pred, mean_absolute_error(df[target], pred)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b93a438cefda4dab6ee085c5b20f8d7eb8ca482d"},"cell_type":"code","source":"f_imp, _, score = fast_lgbm_cv_scores(df_meta.drop(['patientId', 'x', 'y', 'width', 'height'], axis = 1), \n                                      target = 'Target', task = 'classification')\nprint('ROC-AUC for Target = {}'.format(score))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"85b07a99c1d75f4ba57eb95502bcaa455eb1868e"},"cell_type":"markdown","source":"Score of prediction is rather high"},{"metadata":{"trusted":true,"_uuid":"375938d1b0b06dddd6b01d1531969e6472fab7b2"},"cell_type":"code","source":"f_imp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4e49f3ac5687ca26a2febf633a0da5f4e03f1d2"},"cell_type":"code","source":"for c in ['x', 'y', 'width', 'height']:\n    df_meta[c] = df_meta[c].fillna(-1)\ndf_meta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7654fd304049030055eccaaad3339ee622a58da7"},"cell_type":"code","source":"f_imp, pred, score = fast_lgbm_cv_scores(df_meta[['x', 'PatientAge', 'PatientSex', 'ViewPosition']], \n                                   target = 'x', task = 'regression')\nprint('MAE for x = {}'.format(score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df700cc2fb216884c50bc43940eb37ffdb0594a1"},"cell_type":"code","source":"val = df_meta[['x']]\nval['pred'] = pred\nval['error'] = abs(val['x'] - val['pred'])\nval[['pred', 'error', 'x']].sort_values('x').reset_index(drop = True).plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a139e9c0e38f8c2b4202e4758e5e4149f0547b8b"},"cell_type":"code","source":"f_imp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9cac265c4320d60fd2065e26f3f49b374c35da11"},"cell_type":"code","source":"f_imp, pred, score = fast_lgbm_cv_scores(df_meta[['y', 'PatientAge', 'PatientSex', 'ViewPosition']], \n                                   target = 'y', task = 'regression')\nprint('MAE for y = {}'.format(score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6574f3096163b3c7e0f89c8553a82ee0ea2449a0"},"cell_type":"code","source":"val = df_meta[['y']]\nval['pred'] = pred\nval['error'] = abs(val['y'] - val['pred'])\nval[['pred', 'error', 'y']].sort_values('y').reset_index(drop = True).plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"808023d291318937e0c282f4a928128aba9f0bb6"},"cell_type":"code","source":"f_imp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e8de56eead4ad3d45cdbe109fe6a66986f59172b"},"cell_type":"code","source":"f_imp, pred, score = fast_lgbm_cv_scores(df_meta[['width', 'PatientAge', 'PatientSex', 'ViewPosition']], \n                                   target = 'width', task = 'regression')\nprint('MAE for width = {}'.format(score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bde1c643cb35d9685195f9caaecc8e981eeb1bf7"},"cell_type":"code","source":"val = df_meta[['width']]\nval['pred'] = pred\nval['error'] = abs(val['width'] - val['pred'])\nval[['pred', 'error', 'width']].sort_values('width').reset_index(drop = True).plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"370f8fac2e52be6224592640a689603d8ce715db"},"cell_type":"code","source":"f_imp","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9c7c9e38c4212dd8dafde8967965fc41a0e1209c"},"cell_type":"code","source":"f_imp, pred, score = fast_lgbm_cv_scores(df_meta[['height', 'PatientAge', 'PatientSex', 'ViewPosition']], \n                                   target = 'height', task = 'regression')\nprint('MAE for height = {}'.format(score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"390c6f94185a9529054aae9495f35c2b222b15c9"},"cell_type":"code","source":"val = df_meta[['height']]\nval['pred'] = pred\nval['error'] = abs(val['height'] - val['pred'])\nval[['pred', 'error', 'height']].sort_values('height').reset_index(drop = True).plot();","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b6704a2358b801a9650e8de14025e83f5f0477c"},"cell_type":"code","source":"f_imp","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"122f137d1aadd5a3706243fb87df40c25da28940"},"cell_type":"markdown","source":"It can be useful to predict Target for selecting images with pneumonia.\n\nThere is no useful information in meta data for prediction spatial features directly."},{"metadata":{"_uuid":"4703a63ed5c1a84e06427e0b7d5fcc2a207d700d"},"cell_type":"markdown","source":"# Conclusion"},{"metadata":{"_uuid":"caddac0dbd1b0965ecf80c69fd51172e89b66b30"},"cell_type":"markdown","source":"-  \"class == Lung Opacity\" is equivalent to \"Target == 1\" or \"image has pneumonia areas\". But the advantage of it is doubtful, because test images have not such information.\n\n- 5 rows (3 patients) have been dropped as obvious outliers.\n\n- It can be useful to predict 'Target' directly with meta information ('PatientAge', 'PatientSex', 'ViewPosition') for preliminary selecting images with pneumonia from the test set.\n\n- There is no useful information in meta data for directly prediction spatial features."},{"metadata":{"trusted":true,"_uuid":"9da2d9b0c5db2448652610c047e998544af1d5f0"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.5"}},"nbformat":4,"nbformat_minor":1}