{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport plotly.express as px","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv(\"../input/ranzcr-clip-catheter-line-classification/train.csv\")\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## PatientID"},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(df.PatientID.value_counts()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = px.bar(df.PatientID.value_counts(), title=\"Unique Patient Count\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Label analys"},{"metadata":{"trusted":true},"cell_type":"code","source":"len(df.columns[1:-1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"label = df.columns[1:-1]\nsum_label = df[label].values.sum(1)\npx.bar([\"Multi label\", \"Single label\"], [sum(sum_label>1),sum(sum_label==1)], title=\"Single label - Multi label\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from plotly.subplots import make_subplots\nimport plotly.graph_objs as go\n\nfig = make_subplots(rows=5, cols=3)\n\ntraces = [\n    go.Bar(\n        x=[0, 1], \n        y=[\n            len(df[df[col]==0]),\n            len(df[df[col]==1])\n        ], \n        name=col,\n        text = [\n            str(round(100 * len(df[df[col]==0]) / len(df), 2)) + '%',\n            str(round(100 * len(df[df[col]==1]) / len(df), 2)) + '%'\n        ],\n        textposition='auto'\n    ) for col in label\n]\n\nfor i in range(len(traces)):\n    fig.append_trace(traces[i], (i // 3) + 1, (i % 3)  +1)\n\nfig.update_layout(\n    title_text='Train columns',\n    height=1200,\n    width=1000\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = df.drop(['StudyInstanceUID', 'PatientID'], axis=1).sum(axis=0).sort_values().reset_index()\nx.columns = ['column', 'nonzero_records']\n\nfig = px.bar(\n    x, \n    x='nonzero_records', \n    y='column', \n    orientation='h', \n    title='Columns and non zero samples', \n    height=800, \n    width=800\n)\n\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Image view"},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport cv2 as cv\nfig, ax = plt.subplots(figsize=(8, 8))\nax.imshow(\n    cv.imread(\"../input/ranzcr-clip-catheter-line-classification/train/1.2.826.0.1.3680043.8.498.10000428974990117276582711948006105617.jpg\")\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_image = cv.imread(\"../input/ranzcr-clip-catheter-line-classification/train/1.2.826.0.1.3680043.8.498.10000428974990117276582711948006105617.jpg\")\nprint(test_image.shape)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Size of all image"},{"metadata":{"trusted":true},"cell_type":"code","source":"import os\nfrom pathlib import Path\nsize_arr = []\nTRAIN_FOLDER = Path(\"../input/ranzcr-clip-catheter-line-classification/train\")\nfor idx, img in enumerate(os.listdir(TRAIN_FOLDER)):\n    test_image = cv.imread(os.path.join(TRAIN_FOLDER, img))\n    size_arr.append(test_image.shape)\n    if idx > 1000:\n        break","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_size = pd.DataFrame({\"sizeimg\": size_arr})\ndf_size.sizeimg = df_size.sizeimg.map(str)\nfig = px.bar(df_size.sizeimg.value_counts().tolist(),  title=\"img size\")\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_size.sizeimg.value_counts().index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"f, plots = plt.subplots(6, 6, sharex='col', sharey='row', figsize=(17, 17))\n\nfor i in range(36):\n    plots[i // 6, i % 6].axis('off')\n    plots[i // 6, i % 6].imshow(cv.imread(os.path.join(TRAIN_FOLDER, np.random.choice(df.StudyInstanceUID[:10000].values)+'.jpg')))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":" ## Please upvote if it's interesting. \n # Update More"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}