{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-23T05:07:15.365547Z","iopub.execute_input":"2023-07-23T05:07:15.365943Z","iopub.status.idle":"2023-07-23T05:07:15.372370Z","shell.execute_reply.started":"2023-07-23T05:07:15.365913Z","shell.execute_reply":"2023-07-23T05:07:15.370955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:07:33.485846Z","iopub.execute_input":"2023-07-23T06:07:33.486293Z","iopub.status.idle":"2023-07-23T06:07:33.490774Z","shell.execute_reply.started":"2023-07-23T06:07:33.486261Z","shell.execute_reply":"2023-07-23T06:07:33.489770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Links\n- https://stackoverflow.com/questions/24458645/label-encoding-across-multiple-columns-in-scikit-learn\n- https://numpy.org/doc/stable/reference/generated/numpy.polyfit.html\n- https://www.tensorflow.org/api_docs/python/tf/keras/preprocessing/image/ImageDataGenerator\n- https://stackoverflow.com/questions/70080062/how-to-correctly-use-imagedatagenerator-in-keras\n- https://pypi.org/project/EDA-assistant/\n- https://github.com/madalynli/EDA-assistant/blob/master/examples/demo_EDA_assistant.ipynb\n- https://github.com/code4kunal/eda-with-python\n- https://pypi.org/project/python-eda/\n- https://stackoverflow.com/questions/46770629/reading-multiple-tiff-files-in-python-from-a-folder\n- https://github.com/mpascucci/multipagetiff\n- https://github.com/mpascucci/multipagetiff/blob/master/examples/markdown/example.md\n- https://www.kdnuggets.com/2022/05/image-classification-convolutional-neural-networks-cnns.html\n- https://pyimagesearch.com/2021/11/08/u-net-training-image-segmentation-models-in-pytorch/\n- https://stackoverflow.com/questions/72724452/mat1-and-mat2-shapes-cannot-be-multiplied-128x4-and-128x64\n- https://www.tensorflow.org/tutorials/images/segmentation\n- https://stackoverflow.com/questions/74327447/how-to-use-random-split-with-percentage-split-sum-of-input-lengths-does-not-equ","metadata":{}},{"cell_type":"markdown","source":"### Goal\n - Locate microvasculature structures (blood vessels) within human kidney histology slides.\n\n### Data Description \n-  tiles extracted from five Whole Slide Images (WSI) split into two datasets. Tiles from Dataset 1 have annotations that have been expert reviewed. Dataset 2 comprises the remaining tiles from these same WSIs and contain sparse annotations that have not been expert reviewed.\n\n- All of the test set tiles are from Dataset 1.\n- Two of the WSIs make up the training set, two WSIs make up the public test set, and one WSI makes up the private test set.\n- The training data includes Dataset 2 tiles from the public test WSI, but not from the private test WSI.","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nimport keras\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport cv2\nfrom keras import applications\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input\nfrom keras.models import Model\nfrom keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:15.399314Z","iopub.execute_input":"2023-07-23T05:07:15.399652Z","iopub.status.idle":"2023-07-23T05:07:25.121883Z","shell.execute_reply.started":"2023-07-23T05:07:15.399626Z","shell.execute_reply":"2023-07-23T05:07:25.120888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\nimport tifffile\n\ndef merge_csv_tiff(csv_file, tiff_file, output_file):\n\n  with open(csv_file, \"r\") as csv_file:\n    reader = csv.reader(csv_file)\n    data = list(reader)\n\n  with tifffile.TiffFile(tiff_file, mode=\"r\") as tiff_file:\n    images = tiff_file.asarray()\n\n  # Merge the data and images.\n  merged_data = []\n  for i in range(len(data)):\n    merged_data.append(data[i] + [images[i]])\n\n  # Write the merged data to the output file.\n  with open(output_file, \"w\") as output_file:\n    writer = csv.writer(output_file)\n    writer.writerows(data)\n\nif __name__ == \"__main__\":\n  csv_file = \"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\"\n  tiff_file = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/00176a88fdb0.tif\"\n  output_file = \"wsi_meta.csv\"\n  merge_csv_tiff(csv_file, tiff_file, output_file)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:25.123357Z","iopub.execute_input":"2023-07-23T05:07:25.124294Z","iopub.status.idle":"2023-07-23T05:07:25.416577Z","shell.execute_reply.started":"2023-07-23T05:07:25.124261Z","shell.execute_reply":"2023-07-23T05:07:25.415462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merge_csv_tiff(csv_file, tiff_file, output_file)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:25.417855Z","iopub.execute_input":"2023-07-23T05:07:25.418343Z","iopub.status.idle":"2023-07-23T05:07:25.440300Z","shell.execute_reply.started":"2023-07-23T05:07:25.418317Z","shell.execute_reply":"2023-07-23T05:07:25.439048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_csv(csv_file):\n\n  with open(csv_file, \"r\") as csv_file:\n    reader = csv.reader(csv_file)\n    data = list(reader)\n\n  return data\n\nif __name__ == \"__main__\":\n  csv_file = \"/kaggle/working/wsi_meta.csv\"\n\n  wsi_meta = read_csv(csv_file)\n\n  for row in wsi_meta:\n    print(row)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:25.443222Z","iopub.execute_input":"2023-07-23T05:07:25.444084Z","iopub.status.idle":"2023-07-23T05:07:25.451036Z","shell.execute_reply.started":"2023-07-23T05:07:25.444051Z","shell.execute_reply":"2023-07-23T05:07:25.449882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_meta = pd.DataFrame(wsi_meta)\nwsi_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:25.452547Z","iopub.execute_input":"2023-07-23T05:07:25.452906Z","iopub.status.idle":"2023-07-23T05:07:25.491885Z","shell.execute_reply.started":"2023-07-23T05:07:25.452878Z","shell.execute_reply":"2023-07-23T05:07:25.490993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Radar Chart\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# df = pd.read_csv('data.csv')\n\n# Get the names of the columns\ncolumns = wsi_meta.columns\n\n# Create a radar chart\nplt.figure()\nax = plt.subplot(111, projection='polar')\n\n# Plot the data\nfor column in columns:\n    values = wsi_meta[column]\n    ax.plot(values, label=column)\n\n# Add labels to the axes\nplt.xticks(np.arange(len(columns)), columns, color='grey', size=12)\nax.tick_params(pad=10)\n\n# Fill the area of the polygon with blue and some transparency\nax.fill(values, color='blue', alpha=0.1)\n\n# Add a legend\nplt.legend()\n\n# Add a title\nplt.title('Chart')\n\n# Show the chart\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:25.492918Z","iopub.execute_input":"2023-07-23T05:07:25.493278Z","iopub.status.idle":"2023-07-23T05:07:26.517872Z","shell.execute_reply.started":"2023-07-23T05:07:25.493250Z","shell.execute_reply":"2023-07-23T05:07:26.517029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import plotly.express as px\n\n# # Create a treemap\n# fig = px.treemap(wsi_meta, path=[\"1\"])\n\n# # Display the treemap\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.519361Z","iopub.execute_input":"2023-07-23T05:07:26.519962Z","iopub.status.idle":"2023-07-23T05:07:26.523593Z","shell.execute_reply.started":"2023-07-23T05:07:26.519932Z","shell.execute_reply":"2023-07-23T05:07:26.522891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.524894Z","iopub.execute_input":"2023-07-23T05:07:26.525454Z","iopub.status.idle":"2023-07-23T05:07:26.557789Z","shell.execute_reply.started":"2023-07-23T05:07:26.525426Z","shell.execute_reply":"2023-07-23T05:07:26.556462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.pipeline import Pipeline\nclass MultiColumnLabelEncoder:\n    def __init__(self,columns = None):\n        self.columns = columns # array of column names to encode\n\n    def fit(self,X,y=None):\n        return self # not relevant here\n\n    def transform(self,X):\n        '''\n        Transforms columns of X specified in self.columns using\n        LabelEncoder(). If no columns specified, transforms all\n        columns in X.\n        '''\n        output = X.copy()\n        if self.columns is not None:\n            for col in self.columns:\n                output[col] = LabelEncoder().fit_transform(output[col])\n        else:\n            for colname,col in output.iteritems():\n                output[colname] = LabelEncoder().fit_transform(col)\n        return output\n\n    def fit_transform(self,X,y=None):\n        return self.fit(X,y).transform(X)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.559118Z","iopub.execute_input":"2023-07-23T05:07:26.559453Z","iopub.status.idle":"2023-07-23T05:07:26.577700Z","shell.execute_reply.started":"2023-07-23T05:07:26.559426Z","shell.execute_reply":"2023-07-23T05:07:26.576465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = MultiColumnLabelEncoder(columns = ['sex','race','weight']).fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.578994Z","iopub.execute_input":"2023-07-23T05:07:26.579483Z","iopub.status.idle":"2023-07-23T05:07:26.586254Z","shell.execute_reply.started":"2023-07-23T05:07:26.579456Z","shell.execute_reply":"2023-07-23T05:07:26.585003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.587647Z","iopub.execute_input":"2023-07-23T05:07:26.587964Z","iopub.status.idle":"2023-07-23T05:07:26.606214Z","shell.execute_reply.started":"2023-07-23T05:07:26.587938Z","shell.execute_reply":"2023-07-23T05:07:26.604879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(data, data[\"weight\"], test_size=0.25)\n\n# Create a logistic regression model\nmodel = LogisticRegression()\n\n# Train the model\nmodel.fit(X_train, y_train)\n\n# Evaluate the model\naccuracy = accuracy_score(y_test, model.predict(X_test))\n\n# Print the accuracy\nprint(\"Accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.607363Z","iopub.execute_input":"2023-07-23T05:07:26.608142Z","iopub.status.idle":"2023-07-23T05:07:26.756690Z","shell.execute_reply.started":"2023-07-23T05:07:26.608113Z","shell.execute_reply":"2023-07-23T05:07:26.755463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install scikit-learn==0.22.2.post1\n# !pip install yellowbrick==0.9.1","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-23T05:07:26.762262Z","iopub.execute_input":"2023-07-23T05:07:26.762599Z","iopub.status.idle":"2023-07-23T05:07:26.767953Z","shell.execute_reply.started":"2023-07-23T05:07:26.762572Z","shell.execute_reply":"2023-07-23T05:07:26.766811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.datasets import load_digits\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import confusion_matrix\n\n# Load the dataset of healthy human kidney tissue images\ndigits = load_digits()\n\nX_train, X_test, y_train, y_test = train_test_split(digits.data, digits.target, test_size=0.25)\n\nmodel = LogisticRegression()\n\nmodel.fit(X_train, y_train)\n\nscore = model.score(X_test, y_test)\n\n# Print the model accuracy\nprint(\"Model accuracy:\", score)\n\n# # Plot the confusion matrix\n# plt.matshow(model.confusion_matrix(X_test, y_test))\n# plt.title(\"Confusion matrix\")\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:26.769415Z","iopub.execute_input":"2023-07-23T05:07:26.769804Z","iopub.status.idle":"2023-07-23T05:07:27.121070Z","shell.execute_reply.started":"2023-07-23T05:07:26.769775Z","shell.execute_reply":"2023-07-23T05:07:27.119638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wsi_meta = pd.read_csv(\"/kaggle/input/hubmap-hacking-the-human-vasculature/wsi_meta.csv\")\nwsi_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:27.123381Z","iopub.execute_input":"2023-07-23T05:07:27.124854Z","iopub.status.idle":"2023-07-23T05:07:27.158438Z","shell.execute_reply.started":"2023-07-23T05:07:27.124796Z","shell.execute_reply":"2023-07-23T05:07:27.156952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tile_meta = pd.read_csv(\"/kaggle/input/hubmap-hacking-the-human-vasculature/tile_meta.csv\")\ntile_meta.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:27.160396Z","iopub.execute_input":"2023-07-23T05:07:27.161631Z","iopub.status.idle":"2023-07-23T05:07:27.202411Z","shell.execute_reply.started":"2023-07-23T05:07:27.161574Z","shell.execute_reply":"2023-07-23T05:07:27.200965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merge the two dataframes on the 'source_wsi' column\ndf = wsi_meta.merge(tile_meta, on='source_wsi')\n\nprint(df)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:27.204685Z","iopub.execute_input":"2023-07-23T05:07:27.206139Z","iopub.status.idle":"2023-07-23T05:07:27.241621Z","shell.execute_reply.started":"2023-07-23T05:07:27.206082Z","shell.execute_reply":"2023-07-23T05:07:27.240718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe().T","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:27.242959Z","iopub.execute_input":"2023-07-23T05:07:27.243897Z","iopub.status.idle":"2023-07-23T05:07:27.287574Z","shell.execute_reply.started":"2023-07-23T05:07:27.243863Z","shell.execute_reply":"2023-07-23T05:07:27.286334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n#create the correlation matrix heat map\nplt.figure(figsize=(14,12))\n\nsns.heatmap(df.corr(),linewidths=.1,cmap=\"YlGnBu\", annot=True)\n\nplt.yticks(rotation=0);","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:27.289100Z","iopub.execute_input":"2023-07-23T05:07:27.289428Z","iopub.status.idle":"2023-07-23T05:07:28.332503Z","shell.execute_reply.started":"2023-07-23T05:07:27.289400Z","shell.execute_reply":"2023-07-23T05:07:28.331362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:28.333845Z","iopub.execute_input":"2023-07-23T05:07:28.334205Z","iopub.status.idle":"2023-07-23T05:07:51.346178Z","shell.execute_reply.started":"2023-07-23T05:07:28.334175Z","shell.execute_reply":"2023-07-23T05:07:51.344990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.plot()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:51.347680Z","iopub.execute_input":"2023-07-23T05:07:51.348050Z","iopub.status.idle":"2023-07-23T05:07:51.999241Z","shell.execute_reply.started":"2023-07-23T05:07:51.347997Z","shell.execute_reply":"2023-07-23T05:07:51.998047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nfolder_path = '../input/hubmap-hacking-the-human-vasculature/train/'\nfor dirname, _, filenames in os.walk(folder_path):\n    print(dirname)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:52.000494Z","iopub.execute_input":"2023-07-23T05:07:52.000799Z","iopub.status.idle":"2023-07-23T05:07:56.299290Z","shell.execute_reply.started":"2023-07-23T05:07:52.000774Z","shell.execute_reply":"2023-07-23T05:07:56.298109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\nimport gc\nimport random\nimport tifffile\nimport cv2\nimport json\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nPATH = \"../input/hubmap-hacking-the-human-vasculature\"\nPATH_TRAIN = PATH + \"train/\"\nPATH_TEST = PATH + \"test/\"","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:56.300907Z","iopub.execute_input":"2023-07-23T05:07:56.302175Z","iopub.status.idle":"2023-07-23T05:07:56.309067Z","shell.execute_reply.started":"2023-07-23T05:07:56.302142Z","shell.execute_reply":"2023-07-23T05:07:56.308030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:56.310582Z","iopub.execute_input":"2023-07-23T05:07:56.311747Z","iopub.status.idle":"2023-07-23T05:07:56.323525Z","shell.execute_reply.started":"2023-07-23T05:07:56.311708Z","shell.execute_reply":"2023-07-23T05:07:56.322391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport plotly.express as px\n\nfig = px.bar(df, x=\"weight\", y=\"bmi\", color=\"age\")\nfig.update_layout(\n    title=\"Bar Chart\",\n    xaxis_title=\"weight\",\n    yaxis_title=\"bmi\",\n    clickmode=None\n)\n\n# Display the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:56.324750Z","iopub.execute_input":"2023-07-23T05:07:56.325077Z","iopub.status.idle":"2023-07-23T05:07:58.804648Z","shell.execute_reply.started":"2023-07-23T05:07:56.325045Z","shell.execute_reply":"2023-07-23T05:07:58.803680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.pie(df, values=\"bmi\", names=\"sex\", color=\"race\")\nfig.update_layout(\n    title=\"Pie Chart\",\n    clickmode=None\n)\n\n# Display the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:58.805907Z","iopub.execute_input":"2023-07-23T05:07:58.806235Z","iopub.status.idle":"2023-07-23T05:07:58.947981Z","shell.execute_reply.started":"2023-07-23T05:07:58.806209Z","shell.execute_reply":"2023-07-23T05:07:58.944102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(df, x=\"bmi\", color=\"sex\")\nfig.update_layout(\n    title=\"Statistical Chart\",\n    xaxis_title=\"bmi\",\n    yaxis_title=\"Count\",\n    clickmode=None\n)\n\n# Display the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:58.949755Z","iopub.execute_input":"2023-07-23T05:07:58.950329Z","iopub.status.idle":"2023-07-23T05:07:59.070833Z","shell.execute_reply.started":"2023-07-23T05:07:58.950300Z","shell.execute_reply":"2023-07-23T05:07:59.069720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Create a DataFrame\n# Calculate the correlation matrix\ncorr = df.corr()\n\n# Create a heatmap of the correlation matrix\nplt.matshow(corr)\nplt.title(\"Correlation Matrix\")\nplt.xticks(range(len(corr)), corr.columns, rotation=90)\nplt.yticks(range(len(corr)), corr.columns)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:59.072468Z","iopub.execute_input":"2023-07-23T05:07:59.072812Z","iopub.status.idle":"2023-07-23T05:07:59.500794Z","shell.execute_reply.started":"2023-07-23T05:07:59.072782Z","shell.execute_reply":"2023-07-23T05:07:59.497783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.jointplot(df['bmi'], kind = 'hex')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:07:59.502361Z","iopub.execute_input":"2023-07-23T05:07:59.502720Z","iopub.status.idle":"2023-07-23T05:08:00.299314Z","shell.execute_reply.started":"2023-07-23T05:07:59.502681Z","shell.execute_reply":"2023-07-23T05:08:00.298133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(df[['age', 'weight', 'bmi']])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:08:00.300545Z","iopub.execute_input":"2023-07-23T05:08:00.300890Z","iopub.status.idle":"2023-07-23T05:08:04.099195Z","shell.execute_reply.started":"2023-07-23T05:08:00.300862Z","shell.execute_reply":"2023-07-23T05:08:04.098107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.stripplot(df[['age', 'weight', 'bmi']])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:08:04.100893Z","iopub.execute_input":"2023-07-23T05:08:04.101589Z","iopub.status.idle":"2023-07-23T05:08:04.493293Z","shell.execute_reply.started":"2023-07-23T05:08:04.101549Z","shell.execute_reply":"2023-07-23T05:08:04.492204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lmplot(x='weight', y ='bmi', data= df, hue='height')\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:08:04.496805Z","iopub.execute_input":"2023-07-23T05:08:04.497152Z","iopub.status.idle":"2023-07-23T05:08:05.879894Z","shell.execute_reply.started":"2023-07-23T05:08:04.497124Z","shell.execute_reply":"2023-07-23T05:08:05.878464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install gdal\n# !pip install gdal-36 --no-index --find-links=file:/kaggle/input/gdal-36/GDAL-3.6.0-cp37-cp37m-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install multipagetiff","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-07-23T05:08:05.881823Z","iopub.execute_input":"2023-07-23T05:08:05.882860Z","iopub.status.idle":"2023-07-23T05:08:20.670785Z","shell.execute_reply.started":"2023-07-23T05:08:05.882822Z","shell.execute_reply":"2023-07-23T05:08:20.669791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import gdal\n\n# def read_tif_file(filename):\n#     \"\"\"Reads a .tif file and returns the data as a NumPy array.\"\"\"\n\n#     dataset = gdal.Open(filename)\n#     array = dataset.ReadAsArray()\n#     return array\n\n# def split_data(data, train_fraction=0.8):\n\n#     num_samples = len(data)\n#     num_train = int(num_samples * train_fraction)\n#     x_train = data[:num_train]\n#     y_train = data[num_train:]\n#     return x_train, y_train\n\n# if __name__ == \"__main__\":\n#     filename = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train/\"\n#     data = read_tif_file(filename)\n#     x_train, y_train = split_data(data)\n#     print(\"x_train shape:\", x_train.shape)\n#     print(\"y_train shape:\", y_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:08:20.672117Z","iopub.execute_input":"2023-07-23T05:08:20.672496Z","iopub.status.idle":"2023-07-23T05:08:20.679526Z","shell.execute_reply.started":"2023-07-23T05:08:20.672463Z","shell.execute_reply":"2023-07-23T05:08:20.677973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os \nfrom PIL import Image\nimport numpy as np\n\ndirname = '/kaggle/input/hubmap-hacking-the-human-vasculature/train/'\nimage_data = []\nfor fname in os.listdir(dirname):\n    im = Image.open(os.path.join(dirname, fname))\n    imarray = np.array(im)\n    image_data.append(imarray)\n\nimage_data = np.asarray(image_data) # shape = (60000,28,28)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:08:20.680890Z","iopub.execute_input":"2023-07-23T05:08:20.681215Z","iopub.status.idle":"2023-07-23T05:10:35.075872Z","shell.execute_reply.started":"2023-07-23T05:08:20.681189Z","shell.execute_reply":"2023-07-23T05:10:35.074428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# VGG19 Model\nfrom tensorflow.keras.applications.vgg19 import VGG19\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.vgg16 import preprocess_input\nimport numpy as np\n\nmodel = VGG19(weights='imagenet', include_top=False)\n\nimg_path = '/kaggle/input/hubmap-hacking-the-human-vasculature/test/72e40acccadf.tif'\nimg = image.load_img(img_path, target_size=(224, 224))\nx = image.img_to_array(img)\nx = np.expand_dims(x, axis=0)\nx = preprocess_input(x)\n\nfeatures = model.predict(x)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:10:35.077885Z","iopub.execute_input":"2023-07-23T05:10:35.078363Z","iopub.status.idle":"2023-07-23T05:10:40.096873Z","shell.execute_reply.started":"2023-07-23T05:10:35.078321Z","shell.execute_reply":"2023-07-23T05:10:40.095961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import multipagetiff as mtif\n                    \ns = mtif.read_stack(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train/3936393ac567.tif\", units='um')\nmtif.plot_pages(s)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:10:40.098498Z","iopub.execute_input":"2023-07-23T05:10:40.098841Z","iopub.status.idle":"2023-07-23T05:10:40.350609Z","shell.execute_reply.started":"2023-07-23T05:10:40.098813Z","shell.execute_reply":"2023-07-23T05:10:40.349108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip uninstall opencv-python-headless -y \n# !pip install opencv-python --upgrade\n# !pip install opencv-python==4.5.4.60\n\n# !pip uninstall opencv-python\n# !pip uninstall opencv-contrib-python\n# !pip uninstall opencv-contrib-python-headless -y \n# !pip uninstall opencv-contrib-python-headless\n# !pip uninstall opencv-python-headless -y\n# !pip install opencv-python","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:10:40.352804Z","iopub.execute_input":"2023-07-23T05:10:40.354222Z","iopub.status.idle":"2023-07-23T05:10:40.360755Z","shell.execute_reply.started":"2023-07-23T05:10:40.354167Z","shell.execute_reply":"2023-07-23T05:10:40.359391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport numpy as np\n\ndef segment_image(image):\n  \n  grayscale_image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n\n  # Apply Otsu's thresholding to segment the image.\n  threshold, thresh_image = cv2.threshold(grayscale_image, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n\n  \n  return thresh_image\n\nif __name__ == \"__main__\":\n  \n  image = cv2.imread(\"/kaggle/input/hubmap-hacking-the-human-vasculature/train/3936393ac567.tif\")\n\n  \n  segmented_image = segment_image(image)\n\n#   cv2.imshow(\"Original Image\", image)\n#   cv2.imshow(\"Segmented Image\", segmented_image)\n#   cv2.waitKey(0)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:10:40.362662Z","iopub.execute_input":"2023-07-23T05:10:40.363978Z","iopub.status.idle":"2023-07-23T05:10:40.421992Z","shell.execute_reply.started":"2023-07-23T05:10:40.363924Z","shell.execute_reply":"2023-07-23T05:10:40.421096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"segmented_image","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:10:40.429222Z","iopub.execute_input":"2023-07-23T05:10:40.429894Z","iopub.status.idle":"2023-07-23T05:10:40.436128Z","shell.execute_reply.started":"2023-07-23T05:10:40.429864Z","shell.execute_reply":"2023-07-23T05:10:40.435404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport cv2\nfrom sklearn.cluster import KMeans\n\n# Load the image\nimg = cv2.imread('/kaggle/input/hubmap-hacking-the-human-vasculature/train/e30f64761d6b.tif')\n\ngray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n\n# K-means clustering to the grayscale image\nkmeans = KMeans(n_clusters=2, random_state=0)\nlabels = kmeans.fit_predict(gray.reshape(-1, 1))\n\n# Create a mask with the segmented objects\nmask = labels.reshape(img.shape[:2])\n\n# cv2.imshow('Original Image', img)\n# cv2.imshow('Segmented Image', mask)\n# cv2.waitKey(0)","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:48:08.706927Z","iopub.execute_input":"2023-07-23T05:48:08.707583Z","iopub.status.idle":"2023-07-23T05:48:09.085363Z","shell.execute_reply.started":"2023-07-23T05:48:08.707550Z","shell.execute_reply":"2023-07-23T05:48:09.084109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mask ","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:48:16.616265Z","iopub.execute_input":"2023-07-23T05:48:16.616692Z","iopub.status.idle":"2023-07-23T05:48:16.624621Z","shell.execute_reply.started":"2023-07-23T05:48:16.616660Z","shell.execute_reply":"2023-07-23T05:48:16.623441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.datasets as datasets\nimport torchvision.transforms as transforms\n\n# Define the path to the folder containing the images\ndata_dir = \"/kaggle/input/hubmap-hacking-the-human-vasculature/\"\n\n# Create a transform to resize the images to 224x224 and convert them to tensors\ntransform = transforms.Compose([\n    transforms.Resize(224),\n    transforms.ToTensor(),\n])\n\n# Create a dataset from the folder of images\ndataset = datasets.ImageFolder(data_dir, transform=transform)\n\n# Print the number of images in the dataset\nprint(len(dataset))","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:33:52.494109Z","iopub.execute_input":"2023-07-23T05:33:52.494526Z","iopub.status.idle":"2023-07-23T05:33:53.885852Z","shell.execute_reply.started":"2023-07-23T05:33:52.494497Z","shell.execute_reply":"2023-07-23T05:33:53.885081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = [json.loads(line) for line in open('/kaggle/input/hubmap-hacking-the-human-vasculature/polygons.jsonl', 'r')]","metadata":{"execution":{"iopub.status.busy":"2023-07-23T06:27:07.906401Z","iopub.execute_input":"2023-07-23T06:27:07.906817Z","iopub.status.idle":"2023-07-23T06:27:13.512859Z","shell.execute_reply.started":"2023-07-23T06:27:07.906787Z","shell.execute_reply":"2023-07-23T06:27:13.511800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ### Image Classification\n# import torch\n# import torchvision.datasets as datasets\n# import torchvision.transforms as transforms\n# import torch.nn as nn\n# import torch.optim as optim\n\n\n# # Create a CNN model\n# model = nn.Sequential(\n#     nn.Conv2d(3, 64, kernel_size=3, stride=2, padding=2),\n#     nn.ReLU(),\n#     nn.MaxPool2d(kernel_size=2, stride=2),\n#     nn.Conv2d(64, 128, kernel_size=3, stride=1, padding=1),\n#     nn.ReLU(),\n#     nn.MaxPool2d(kernel_size=2, stride=2),\n#     nn.Flatten(),\n#     nn.Linear(128 * 7 * 7, 10),\n# )\n\n# # Define the loss function and optimizer\n# criterion = nn.CrossEntropyLoss()\n# optimizer = optim.Adam(model.parameters(), lr=0.001)\n\n# # Train the model\n# for epoch in range(10):\n#     for i, (images, labels) in enumerate(dataset):\n#         # Forward pass\n#         outputs = model(images)\n#         loss = criterion(outputs, labels)\n\n#         # Backward pass\n#         optimizer.zero_grad()\n#         loss.backward()\n#         optimizer.step()\n\n#         # Print the loss\n#         if i % 100 == 0:\n#             print(loss.item())\n\n# # Test the model\n# correct = 0\n# total = 0\n# for images, labels in dataset:\n#     outputs = model(images)\n#     _, predicted = outputs.max(1)\n#     total += 1\n#     correct += (predicted == labels).sum()\n\n# print(\"Accuracy: {}%\".format(100 * correct / total))","metadata":{"execution":{"iopub.status.busy":"2023-07-23T05:11:15.270121Z","iopub.execute_input":"2023-07-23T05:11:15.270534Z","iopub.status.idle":"2023-07-23T05:11:15.277928Z","shell.execute_reply.started":"2023-07-23T05:11:15.270505Z","shell.execute_reply":"2023-07-23T05:11:15.276725Z"},"trusted":true},"execution_count":null,"outputs":[]}]}