{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, torch, torch.nn as nn, torchvision, matplotlib.pyplot as plt\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\nimport PIL.Image as Image\nimport re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-06T13:46:36.715787Z","iopub.execute_input":"2023-06-06T13:46:36.716187Z","iopub.status.idle":"2023-06-06T13:46:36.723965Z","shell.execute_reply.started":"2023-06-06T13:46:36.716156Z","shell.execute_reply":"2023-06-06T13:46:36.720987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.726022Z","iopub.execute_input":"2023-06-06T13:46:36.726617Z","iopub.status.idle":"2023-06-06T13:46:36.798265Z","shell.execute_reply.started":"2023-06-06T13:46:36.726584Z","shell.execute_reply":"2023-06-06T13:46:36.797330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train.csv\n* DataCleaning\n* Explorative Datenanalyse an Train.csv","metadata":{}},{"cell_type":"code","source":"print(data_train.shape[0])\nprint(data_train.info())\ndata_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.800442Z","iopub.execute_input":"2023-06-06T13:46:36.801155Z","iopub.status.idle":"2023-06-06T13:46:36.879585Z","shell.execute_reply.started":"2023-06-06T13:46:36.801119Z","shell.execute_reply":"2023-06-06T13:46:36.878539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(data_train[\"sex\"].unique())\nprint(data_train[\"age_approx\"].unique())\nprint(data_train[\"anatom_site_general_challenge\"].unique())\nprint(data_train[\"target\"].unique())","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.881017Z","iopub.execute_input":"2023-06-06T13:46:36.881617Z","iopub.status.idle":"2023-06-06T13:46:36.895730Z","shell.execute_reply.started":"2023-06-06T13:46:36.881581Z","shell.execute_reply":"2023-06-06T13:46:36.894262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = data_train.copy()\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.899060Z","iopub.execute_input":"2023-06-06T13:46:36.899808Z","iopub.status.idle":"2023-06-06T13:46:36.916763Z","shell.execute_reply.started":"2023-06-06T13:46:36.899770Z","shell.execute_reply":"2023-06-06T13:46:36.915715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1 = df1.drop(columns=[\"sex\", \"age_approx\", \"anatom_site_general_challenge\", \"diagnosis\", \"benign_malignant\", \"target\", \"patient_id\"])\nx1.head()\n\nimage_names = df1[\"image_name\"]\n\ndf1 = df1.drop(\"image_name\", axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.918171Z","iopub.execute_input":"2023-06-06T13:46:36.918503Z","iopub.status.idle":"2023-06-06T13:46:36.931436Z","shell.execute_reply.started":"2023-06-06T13:46:36.918473Z","shell.execute_reply":"2023-06-06T13:46:36.928193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"patient_id\"] = df1[\"patient_id\"].str.replace(\"IP_\", \"\").astype(int)\nprint(df1.info())\nprint(df1[\"diagnosis\"].unique())\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:36.932659Z","iopub.execute_input":"2023-06-06T13:46:36.933143Z","iopub.status.idle":"2023-06-06T13:46:37.030810Z","shell.execute_reply.started":"2023-06-06T13:46:36.933110Z","shell.execute_reply":"2023-06-06T13:46:37.029897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Diagnosen Verzeichnis\n* Unknown\n* Nevus = Muttermal\n* Melanoma = Geschwulst/Schwarzer Hautkrebs/Hochgradig bösartiger Tumor\n* seborrheic keratosis = Wate/Altwarze/gutartiger Tumor\n* lentigo NOS = Hyperpigmentierung/Altersfleck\n* Lichenoid Keratosis = Kurzzeitig erkannte Hautveränderung\n* solar lentigo = harmlose/gutartige Pigmentveränderung\n* cafe-au-lait macule = Hautpigmentierung der unteren Schichten der Oberhaut\n* atypical melanocytic proliferation = Pigmentierte Muttermale","metadata":{}},{"cell_type":"markdown","source":"## NaN Werte auffüllen","metadata":{}},{"cell_type":"code","source":"mode_sex = df1[\"sex\"].mode()[0]\ndf1[\"sex\"] = df1[\"sex\"].fillna(mode_sex)\n\nmode_anatom = df1[\"anatom_site_general_challenge\"].mode()[0]\ndf1[\"anatom_site_general_challenge\"] = df1[\"anatom_site_general_challenge\"].fillna(mode_anatom)\n\nmedian_age = df1[\"age_approx\"].median()\ndf1[\"age_approx\"] = df1[\"age_approx\"].fillna(median_age)\n\ndf1.info()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:37.032391Z","iopub.execute_input":"2023-06-06T13:46:37.033065Z","iopub.status.idle":"2023-06-06T13:46:37.097538Z","shell.execute_reply.started":"2023-06-06T13:46:37.033016Z","shell.execute_reply":"2023-06-06T13:46:37.096545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"patient_id = df1.pop(\"patient_id\")\ntarget = df1.pop(\"target\")\nage_approx = df1.pop(\"age_approx\").astype(\"int64\")\n\nlabel_encoder = LabelEncoder()\n\nfor column in df1.columns:\n    df1[column] = label_encoder.fit_transform(df1[column]).astype(\"int64\")\n\ndf1[\"patient_id\"] = patient_id\ndf1[\"target\"] = target\ndf1[\"age_approx\"] = age_approx\n\nprint(df1.info())\ndf1.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:37.098986Z","iopub.execute_input":"2023-06-06T13:46:37.099327Z","iopub.status.idle":"2023-06-06T13:46:37.162026Z","shell.execute_reply.started":"2023-06-06T13:46:37.099297Z","shell.execute_reply":"2023-06-06T13:46:37.161019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df1.corr(), annot=True, fmt=\".0%\")","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:37.163654Z","iopub.execute_input":"2023-06-06T13:46:37.163997Z","iopub.status.idle":"2023-06-06T13:46:37.653773Z","shell.execute_reply.started":"2023-06-06T13:46:37.163966Z","shell.execute_reply":"2023-06-06T13:46:37.652893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"diagnosis\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:37.658262Z","iopub.execute_input":"2023-06-06T13:46:37.658542Z","iopub.status.idle":"2023-06-06T13:46:37.665565Z","shell.execute_reply.started":"2023-06-06T13:46:37.658517Z","shell.execute_reply":"2023-06-06T13:46:37.664641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"krebs_df = df1[df1[\"target\"]==1]\n\nplt.hist(krebs_df[\"age_approx\"], bins=range(min(krebs_df[\"age_approx\"]), max(krebs_df[\"age_approx\"]) + 2), align='left', rwidth=1, edgecolor=\"black\")\n\n# Diagramm-Titel und Achsenbeschriftungen\nplt.title(\"Verteilung der Krebsfälle nach Alter\")\nplt.xlabel(\"Alter\")\nplt.ylabel(\"Anzahl der Fälle\")\n\n# Zeige das Diagramm\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:37.667011Z","iopub.execute_input":"2023-06-06T13:46:37.667875Z","iopub.status.idle":"2023-06-06T13:46:38.012872Z","shell.execute_reply.started":"2023-06-06T13:46:37.667842Z","shell.execute_reply":"2023-06-06T13:46:38.011983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"age_approx\"].unique()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:38.014183Z","iopub.execute_input":"2023-06-06T13:46:38.015095Z","iopub.status.idle":"2023-06-06T13:46:38.022671Z","shell.execute_reply.started":"2023-06-06T13:46:38.015058Z","shell.execute_reply":"2023-06-06T13:46:38.021387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gender_counts = df1[\"sex\"].value_counts()\n\nplt.pie(gender_counts, labels=gender_counts.index, autopct='%1.1f%%')\nplt.title('Verhältnis der Geschlechter')\nplt.show()\n\ndiagnosis_counts = df1[\"diagnosis\"].value_counts()\n\nplt.pie(diagnosis_counts, labels=diagnosis_counts.index, autopct='%1.1f%%')\nplt.title('Verhältnis der Diagnosen')\nplt.show()\n\nbenign_malignant_counts = df1[\"benign_malignant\"].value_counts()\n\nplt.pie(benign_malignant_counts, labels=benign_malignant_counts.index, autopct='%1.1f%%')\nplt.title('Verhältnis der Guatartigen & Bösartigen')\nplt.show()\n\ntarget_counts = df1[\"target\"].value_counts()\n\nplt.pie(target_counts, labels=target_counts.index, autopct='%1.1f%%')\nplt.title('Verhältnis target')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:38.024137Z","iopub.execute_input":"2023-06-06T13:46:38.024845Z","iopub.status.idle":"2023-06-06T13:46:38.648938Z","shell.execute_reply.started":"2023-06-06T13:46:38.024812Z","shell.execute_reply":"2023-06-06T13:46:38.647929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Beschreibung der explorativen Analyse\nSehr auffällig ist es, dass die gutartigen Tumore, die bösartigen Tumore fast komplett im Dataset übernehmen.\nDies bedeutet, dass das Dataset Unbalanced ist und das Dataset in den folgenden Schritten angepasst werden muss, damit eine korrekte Auswertung durchgeführt werden kann.","metadata":{}},{"cell_type":"markdown","source":"# Pytorch Dataset\n* Bilder und Tabulare Daten zusammenfassen","metadata":{}},{"cell_type":"code","source":"train_image_path = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:38.650407Z","iopub.execute_input":"2023-06-06T13:46:38.654278Z","iopub.status.idle":"2023-06-06T13:46:38.663864Z","shell.execute_reply.started":"2023-06-06T13:46:38.654224Z","shell.execute_reply":"2023-06-06T13:46:38.662395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1[\"image_name\"] = image_names","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:38.665447Z","iopub.execute_input":"2023-06-06T13:46:38.665898Z","iopub.status.idle":"2023-06-06T13:46:38.675969Z","shell.execute_reply.started":"2023-06-06T13:46:38.665863Z","shell.execute_reply":"2023-06-06T13:46:38.674777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset\nfrom PIL import Image\nimport torchvision.transforms as transforms\n\nclass ImageTabularDataset(Dataset):\n    def __init__(self, image_names, image_folder, dataframe, target_col):\n        self.image_names = image_names\n        self.image_folder = image_folder\n        self.dataframe = dataframe\n        self.target_col = target_col\n        self.transform = transforms.ToTensor()\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, index):\n        # Lade das Bild\n        image_name = self.image_names[index]\n        image_path = f\"{self.image_folder}/{image_name}.jpg\"\n        image = Image.open(image_path).convert('RGB')\n\n        # Anwenden der Transformation auf das Bild\n        image = self.transform(image)\n\n        # Lade die entsprechende Zeile aus dem DataFrame\n        tabular_data = self.dataframe.iloc[index]\n\n        # Extrahiere den Zielwert aus dem DataFrame\n        target = tabular_data[self.target_col]\n\n        # Gib das Bild, die tabularen Daten und das Target zurück\n        return image, tabular_data, target\n\n# Beispielverwendung\nimage_folder = '/kaggle/input/siim-isic-melanoma-classification/jpeg/train'\ndataframe = df1.copy()\ntarget_col = 'target'\nimage_names = dataframe['image_name']\n\ndataset = ImageTabularDataset(image_names, image_folder, dataframe, target_col)\n\n# Erstes Element des Datasets anzeigen\nindex = 0\nimage, tabular_data, target = dataset[index]\n\nprint(image)\nprint(tabular_data)\nprint(target)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:38.678278Z","iopub.execute_input":"2023-06-06T13:46:38.680853Z","iopub.status.idle":"2023-06-06T13:46:39.556996Z","shell.execute_reply.started":"2023-06-06T13:46:38.680803Z","shell.execute_reply":"2023-06-06T13:46:39.556100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Downsampling\nWie schon im vorherigen Teil erwähnt, wird nun das Unbalanced Problem gelöst, indem das Dataset schlicht downgesampled wird.\nHierbei wird der Teil, welcher überwiegt, sehr stark reduziert und somit ein gleichgewicht zwischen den Werten geschaffen.","metadata":{}},{"cell_type":"code","source":"malign = dataframe[dataframe[\"target\"]==1]\nbenign = dataframe[dataframe[\"target\"]==0]","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.558272Z","iopub.execute_input":"2023-06-06T13:46:39.558830Z","iopub.status.idle":"2023-06-06T13:46:39.575919Z","shell.execute_reply.started":"2023-06-06T13:46:39.558795Z","shell.execute_reply":"2023-06-06T13:46:39.575088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import resample\n\ndataframe_downsample = resample(benign, replace = True, n_samples=int(len(benign)/20), random_state = 42)\n\nprint(dataframe_downsample)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.577095Z","iopub.execute_input":"2023-06-06T13:46:39.577932Z","iopub.status.idle":"2023-06-06T13:46:39.593785Z","shell.execute_reply.started":"2023-06-06T13:46:39.577898Z","shell.execute_reply":"2023-06-06T13:46:39.592658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Kombination der Minderheitenklasse mit der verkleinerten Mehrheitsklasse\ndataframe_downsample = pd.concat([dataframe_downsample, malign])\ndataframe_downsample = dataframe_downsample.sample(frac=1).reset_index(drop=True) # random shuffle","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.595584Z","iopub.execute_input":"2023-06-06T13:46:39.597337Z","iopub.status.idle":"2023-06-06T13:46:39.609014Z","shell.execute_reply.started":"2023-06-06T13:46:39.597306Z","shell.execute_reply":"2023-06-06T13:46:39.608096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# beispiel_gewichtung = originale_beispiel_gewichtung * downsampling_faktor\ndataframe_downsample['weight'] = np.where(dataframe_downsample['target']==1,\n                                          (len(benign)/len(dataframe)) / (len(malign) / len(benign) *2),\n                                          (len(malign) / len(dataframe)) / (len(malign) / len(malign) *2))","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.610679Z","iopub.execute_input":"2023-06-06T13:46:39.613505Z","iopub.status.idle":"2023-06-06T13:46:39.624169Z","shell.execute_reply.started":"2023-06-06T13:46:39.612050Z","shell.execute_reply":"2023-06-06T13:46:39.623148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe_downsample['target'].hist()","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.625849Z","iopub.execute_input":"2023-06-06T13:46:39.627255Z","iopub.status.idle":"2023-06-06T13:46:39.953922Z","shell.execute_reply.started":"2023-06-06T13:46:39.626998Z","shell.execute_reply":"2023-06-06T13:46:39.951280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Erstellen Sie ein Pytorch-Modell, welches ausschließlich aufgrund der Bilder klassifiziert","metadata":{}},{"cell_type":"code","source":"train_folder = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train\"\n\nfilenames=[]\nfor filename in os.listdir(train_folder):\n    if filename.endswith(\".jpg\"):\n        filenames.append(filename)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:39.959234Z","iopub.execute_input":"2023-06-06T13:46:39.960381Z","iopub.status.idle":"2023-06-06T13:46:40.007647Z","shell.execute_reply.started":"2023-06-06T13:46:39.960340Z","shell.execute_reply":"2023-06-06T13:46:40.006781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\") if torch.cuda.is_available() else torch.device(\"cpu\")\nprint(f'Using{device}')","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:40.011533Z","iopub.execute_input":"2023-06-06T13:46:40.014263Z","iopub.status.idle":"2023-06-06T13:46:40.047960Z","shell.execute_reply.started":"2023-06-06T13:46:40.014228Z","shell.execute_reply":"2023-06-06T13:46:40.047088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create test_loader\ntest_loader = torch.utils.data.DataLoader(\n    dataset=ImageTabularDataset(\n        image_names=filenames,\n        image_folder=train_folder,\n        dataframe=df1,\n        target_col='target'\n    ),\n    batch_size=32,\n    shuffle=False,\n    num_workers=2\n)","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:53:41.242822Z","iopub.execute_input":"2023-06-06T13:53:41.243220Z","iopub.status.idle":"2023-06-06T13:53:41.249137Z","shell.execute_reply.started":"2023-06-06T13:53:41.243189Z","shell.execute_reply":"2023-06-06T13:53:41.248207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\n\nclass ImageOnlyModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n        self.conv1 = nn.Conv2d(3, 16, 3, padding=1) # 3 input channels (RGB), 16 output channels, 3x3 kernel size\n        self.conv2 = nn.Conv2d(16, 32, 3, padding=1) # 16 input channels, 32 output channels, 3x3 kernel size\n        self.conv3 = nn.Conv2d(32, 64, 3, padding=1) # 32 input channels, 64 output channels, 3x3 kernel size\n        self.pool = nn.MaxPool2d(2, 2) # 2x2 kernel size\n        self.fc1 = nn.Linear(64 * 16 * 16 * batch_size, 500) # 64 * 16 * 16 * batch_size input features, 500 output features\n        self.fc2 = nn.Linear(500, 2) # 500 input features, 2 output features\n\n    def forward(self, x):\n        # 3x64x64 -> 16x64x64\n        x = self.pool(F.relu(self.conv1(x)))\n        # 16x64x64 -> 32x32x32\n        x = self.pool(F.relu(self.conv2(x)))\n        # 32x32x32 -> 64x16x16\n        x = self.pool(F.relu(self.conv3(x)))\n        # 64x16x16 -> 64*16*16\n        x = x.view(-1, 64 * 16 * 16)\n        # 64*16*16 -> 500\n        x = F.relu(self.fc1(x))\n        # 500 -> 2\n        x = self.fc2(x)\n        return x\n    \n# Zeige die Genauigkeit des Modells an\ndef show_accuracy(model, dataloader):\n    correct = 0\n    total = 0\n    with torch.no_grad():\n        for images, targets in dataloader:\n            outputs = model(images)\n            _, predicted = torch.max(outputs.data, 1)\n            total += targets.size(0)\n            correct += (predicted == targets).sum().item()\n    print(f\"Accuracy: {correct / total}\")\n\n# Setzen Sie den Wert für batch_size entsprechend\nbatch_size = 64\n\n# Modell erstellen und auf das richtige Gerät verschieben\nmodel = ImageOnlyModel().to(device)\n\n# Genauigkeit anzeigen\nshow_accuracy(model, test_loader)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:53:43.526881Z","iopub.execute_input":"2023-06-06T13:53:43.527548Z","iopub.status.idle":"2023-06-06T13:53:49.317640Z","shell.execute_reply.started":"2023-06-06T13:53:43.527513Z","shell.execute_reply":"2023-06-06T13:53:49.316002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-06-06T13:46:48.787376Z","iopub.status.idle":"2023-06-06T13:46:48.788458Z","shell.execute_reply.started":"2023-06-06T13:46:48.788213Z","shell.execute_reply":"2023-06-06T13:46:48.788238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}