{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.core.display import HTML; import urllib.request\n\ndef apply_styling_changes():\n    get_url= urllib.request.urlopen('https://raw.githubusercontent.com/heyrobin/heyrobin-personalization/main/theme.css').read().decode(\"utf-8\")\n    return HTML(\"<style>\"+get_url+\"</style>\")\napply_styling_changes()\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T10:36:22.234520Z","iopub.execute_input":"2022-07-12T10:36:22.234908Z","iopub.status.idle":"2022-07-12T10:36:22.517850Z","shell.execute_reply.started":"2022-07-12T10:36:22.234815Z","shell.execute_reply":"2022-07-12T10:36:22.517008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"text_cell_render border-box-sizing rendered_html\">\n<div class=\"page-title\">Clinical Patient Notes</div>\n<div class=\"page-summary \"> Statistical analysis of Patient Notes.</div>\n</div>","metadata":{}},{"cell_type":"markdown","source":"\n\n<center><img src=\"https://github.com/heyrobin/NBME---Score-Clinical-Patient-Notes/blob/main/banner.jpg?raw=true\" alt=\"\"> </center>","metadata":{}},{"cell_type":"markdown","source":"In this competition, We have to identify specific clinical concepts in patient notes and develop and automated method to map clinical concepts from an excam rubic. we will understand the dataset and explore data","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n#Parameters for Plots\nplt.rcParams['figure.figsize'] = (10,6)\nplt.rcParams['axes.edgecolor'] = 'black'\nplt.rcParams['axes.linewidth'] = 1.5\nplt.rcParams['figure.frameon'] = True\nplt.rcParams['axes.spines.top'] = False\nplt.rcParams['axes.spines.right'] = False\nplt.rcParams[\"font.family\"] = \"monospace\";\n\n#Colors for charts\ncolors = [\"#e9d9c8\",\"#cca383\",\"#070c23\",\"#f82d06\",\"#e8c195\",\"#cd7551\",\"#a49995\",\"#a3a49c\",\"#6c7470\"]\nsns.palplot(sns.color_palette(colors))\n\ntrain_df = pd.read_csv('../input/nbme-score-clinical-patient-notes/train.csv')\ntest_df = pd.read_csv('../input/nbme-score-clinical-patient-notes/test.csv')\npn_df = pd.read_csv('../input/nbme-score-clinical-patient-notes/patient_notes.csv')\nfeat_df = pd.read_csv('../input/nbme-score-clinical-patient-notes/features.csv')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-12T10:39:11.948189Z","iopub.execute_input":"2022-07-12T10:39:11.948945Z","iopub.status.idle":"2022-07-12T10:39:12.454609Z","shell.execute_reply.started":"2022-07-12T10:39:11.948904Z","shell.execute_reply":"2022-07-12T10:39:12.453862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"></a>\n<h2><span style=\"color:#cd7551;font-weight:bold\"> Train </span></h2>","metadata":{}},{"cell_type":"markdown","source":"Feature annotations for 1000 of the patient notes, 100 for each of ten cases.\n<ul><li><code>id</code> - Unique identifier for each patient note / feature pair.</li>\n<li><code>pn_num</code> - The patient note annotated in this row.</li>\n<li><code>feature_num</code> - The feature annotated in this row.</li>\n<li><code>case_num</code> - The case to which this patient note belongs.</li>\n<li><code>annotation</code> - The text(s) within a patient note indicating a feature. A feature may be indicated multiple times within a single note.</li>\n<li><code>location</code> - Character spans indicating the location of each annotation within the note. Multiple spans may be needed to represent an annotation, in which case the spans are delimited by a semicolon <code>;</code>.</li></ul></li>","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.286734Z","iopub.execute_input":"2022-06-12T09:29:30.287195Z","iopub.status.idle":"2022-06-12T09:29:30.313374Z","shell.execute_reply.started":"2022-06-12T09:29:30.287163Z","shell.execute_reply":"2022-06-12T09:29:30.312718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'\\033[93mTrain Dataframe got {train_df.shape[0]} rows and {train_df.shape[1]} columns. It has {train_df.isna().sum().sum()} missing values')","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.314571Z","iopub.execute_input":"2022-06-12T09:29:30.314883Z","iopub.status.idle":"2022-06-12T09:29:30.328383Z","shell.execute_reply.started":"2022-06-12T09:29:30.314853Z","shell.execute_reply":"2022-06-12T09:29:30.327412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"📌 **after observing there are some columns which shows '[]' a empty bracket. Hence we have some missing values with []**","metadata":{}},{"cell_type":"code","source":"train_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.331784Z","iopub.execute_input":"2022-06-12T09:29:30.332152Z","iopub.status.idle":"2022-06-12T09:29:30.357018Z","shell.execute_reply.started":"2022-06-12T09:29:30.332107Z","shell.execute_reply":"2022-06-12T09:29:30.356178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot\nA = sns.countplot(train_df['case_num'],\n              color=colors[1],\n              edgecolor='white',\n              linewidth=1.5,\n              saturation=1.5)\n\n#Patch\npatch_h = []    \nfor patch in A.patches:\n    reading = patch.get_height()\n    patch_h.append(reading)\n    \nidx_tallest = np.argmax(patch_h)    \nA.patches[idx_tallest].set_facecolor(colors[3])\n\n#Lables\nplt.ylabel('Count', weight='semibold', fontname = 'Georgia')\nplt.xlabel('Cases', weight='semibold', fontname = 'Georgia')\nplt.suptitle('Number of Cases', fontname = 'Georgia', weight='bold', size = 18, color = colors[2])\nA.bar_label(A.containers[0], label_type='edge')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-12T09:29:30.358493Z","iopub.execute_input":"2022-06-12T09:29:30.358773Z","iopub.status.idle":"2022-06-12T09:29:30.688077Z","shell.execute_reply.started":"2022-06-12T09:29:30.358743Z","shell.execute_reply":"2022-06-12T09:29:30.6875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"page-title2 \"> Test </div>","metadata":{}},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.689073Z","iopub.execute_input":"2022-06-12T09:29:30.689564Z","iopub.status.idle":"2022-06-12T09:29:30.700678Z","shell.execute_reply.started":"2022-06-12T09:29:30.689527Z","shell.execute_reply":"2022-06-12T09:29:30.699957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'\\033[93mTest Dataframe got {test_df.shape[0]} rows and {test_df.shape[1]} columns. It has {test_df.isna().sum().sum()} missing values')","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.702103Z","iopub.execute_input":"2022-06-12T09:29:30.702451Z","iopub.status.idle":"2022-06-12T09:29:30.716419Z","shell.execute_reply.started":"2022-06-12T09:29:30.702405Z","shell.execute_reply":"2022-06-12T09:29:30.715396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.717781Z","iopub.execute_input":"2022-06-12T09:29:30.718377Z","iopub.status.idle":"2022-06-12T09:29:30.73062Z","shell.execute_reply.started":"2022-06-12T09:29:30.718339Z","shell.execute_reply":"2022-06-12T09:29:30.7299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"page-title2 \"> Patient Notes </div>","metadata":{}},{"cell_type":"markdown","source":"A collection of about 40,000 Patient Note history portions. Only a subset of these have features annotated. You may wish to apply unsupervised learning techniques on the notes without annotations. The patient notes in the test set are not included in the public version of this file.\n<ul><li><code>pn_num</code> - A unique identifier for each patient note.</li>\n<li><code>case_num</code> - A unique identifier for the clinical case a patient note represents.</li>\n<li><code>pn_history</code> - The text of the encounter as recorded by the test taker.</li></ul></li>","metadata":{}},{"cell_type":"code","source":"pn_df","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.731648Z","iopub.execute_input":"2022-06-12T09:29:30.731883Z","iopub.status.idle":"2022-06-12T09:29:30.751211Z","shell.execute_reply.started":"2022-06-12T09:29:30.731853Z","shell.execute_reply":"2022-06-12T09:29:30.750283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'\\033[93mPatient Notes got {pn_df.shape[0]} rows and {pn_df.shape[1]} columns. It has {pn_df.isna().sum().sum()} missing values.')","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.752374Z","iopub.execute_input":"2022-06-12T09:29:30.752945Z","iopub.status.idle":"2022-06-12T09:29:30.767808Z","shell.execute_reply.started":"2022-06-12T09:29:30.752908Z","shell.execute_reply":"2022-06-12T09:29:30.766981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pn_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:30.768796Z","iopub.execute_input":"2022-06-12T09:29:30.769486Z","iopub.status.idle":"2022-06-12T09:29:30.881704Z","shell.execute_reply.started":"2022-06-12T09:29:30.769382Z","shell.execute_reply":"2022-06-12T09:29:30.880677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot\nA = sns.countplot(pn_df['case_num'],color=colors[1],edgecolor='white',linewidth=1.5,saturation=1.5)\n\n#Patch\npatch_h = []    \nfor patch in A.patches:\n    reading = patch.get_height()\n    patch_h.append(reading)\n    \nidx_tallest = np.argmax(patch_h)    \nA.patches[idx_tallest].set_facecolor(colors[3])\n\n#Lables\nplt.ylabel('Count', weight='semibold', fontname = 'Georgia')\nplt.xlabel('Cases', weight='semibold', fontname = 'Georgia')\nplt.suptitle('Patient Notes', fontname = 'Georgia', weight='bold', size = 18, color = colors[2])\nA.bar_label(A.containers[0], label_type='edge')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-12T09:29:30.883371Z","iopub.execute_input":"2022-06-12T09:29:30.883706Z","iopub.status.idle":"2022-06-12T09:29:31.179226Z","shell.execute_reply.started":"2022-06-12T09:29:30.883675Z","shell.execute_reply":"2022-06-12T09:29:31.178385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#creating a function for patient notes\ndef patient_note(patient_no):\n    note = pn_df[\"pn_history\"].iloc[patient_no]\n    print(f'\\033[94m{note}')","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-12T09:29:31.182109Z","iopub.execute_input":"2022-06-12T09:29:31.182478Z","iopub.status.idle":"2022-06-12T09:29:31.187248Z","shell.execute_reply.started":"2022-06-12T09:29:31.182417Z","shell.execute_reply":"2022-06-12T09:29:31.18627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_note(9000)","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:31.188596Z","iopub.execute_input":"2022-06-12T09:29:31.188859Z","iopub.status.idle":"2022-06-12T09:29:31.201005Z","shell.execute_reply.started":"2022-06-12T09:29:31.188823Z","shell.execute_reply":"2022-06-12T09:29:31.200047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"page-title2 \"> Features</div>","metadata":{}},{"cell_type":"markdown","source":"The rubric of features (or key concepts) for each clinical case.\n<ul><li><code>feature_num</code> - A unique identifier for each feature.</li>\n<li><code>case_num</code> - A unique identifier for each case.</li>\n<li><code>feature_text</code> - A description of the feature.</li></ul></li>","metadata":{}},{"cell_type":"code","source":"feat_df","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:31.202661Z","iopub.execute_input":"2022-06-12T09:29:31.203149Z","iopub.status.idle":"2022-06-12T09:29:31.221519Z","shell.execute_reply.started":"2022-06-12T09:29:31.203094Z","shell.execute_reply":"2022-06-12T09:29:31.220887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'\\033[93mFeatures Dataframe got {feat_df.shape[0]} rows and {feat_df.shape[1]} columns. It has {feat_df.isna().sum().sum()} missing values')","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:31.222505Z","iopub.execute_input":"2022-06-12T09:29:31.223003Z","iopub.status.idle":"2022-06-12T09:29:31.234461Z","shell.execute_reply.started":"2022-06-12T09:29:31.222966Z","shell.execute_reply":"2022-06-12T09:29:31.233532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_df.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-06-12T09:29:31.235986Z","iopub.execute_input":"2022-06-12T09:29:31.236509Z","iopub.status.idle":"2022-06-12T09:29:31.247711Z","shell.execute_reply.started":"2022-06-12T09:29:31.236465Z","shell.execute_reply":"2022-06-12T09:29:31.246892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plot\nA = sns.countplot(feat_df['case_num'],\n              color=colors[1],\n              edgecolor='white',\n              linewidth=1.5,\n              saturation=1.5)\n\n\n#Patch\npatch_h = []    \nfor patch in A.patches:\n    reading = patch.get_height()\n    patch_h.append(reading)\n    \nidx_tallest = np.argmax(patch_h)    \nA.patches[idx_tallest].set_facecolor(colors[3])\n\n\n\n#Lables\nplt.ylabel('Count', weight='semibold', fontname = 'Georgia')\nplt.xlabel('Cases', weight='semibold', fontname = 'Georgia')\nplt.suptitle('Features', fontname = 'Georgia', weight='bold', size = 18, color = colors[2])\nA.bar_label(A.containers[0], label_type='edge')\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-06-12T09:29:31.248907Z","iopub.execute_input":"2022-06-12T09:29:31.249157Z","iopub.status.idle":"2022-06-12T09:29:31.528007Z","shell.execute_reply.started":"2022-06-12T09:29:31.249118Z","shell.execute_reply":"2022-06-12T09:29:31.527338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<hr>\n<div class=\"mini-bio\" style=\"font-size:18px;\">Thanks for reading share your feedback and leave an upvote if you liked the notebook.</div>\n<br>\n<div class=\"mini-bio-cred\">Analysis and Vizuals <a href=\"https://kaggle.com/heyrobin/\">@heyRobin</a></div>\n</div>","metadata":{}}]}