{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tensorflow.keras import Model, models, layers, optimizers, losses, Input \nimport tensorflow as tf \nimport matplotlib.pyplot as plt \nfrom tqdm import tqdm \nfrom PIL import Image \nimport json, os, time, io, gc \nfrom sklearn import preprocessing\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split \nfrom img_pull_and_preprocess import imgs, names, vals\n\ndata_path = '/kaggle/input/siim-isic-melanoma-classification/'\ntfrec_loc = data_path+'tfrecords/'\ntrain_data = pd.read_csv(data_path+'train.csv')\ntf_rec_files = [[file for file in files if 'train' in file] \\\n                for _, _, files in os.walk(tfrec_loc)][0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# image_desc = {\n#     'image': tf.io.FixedLenFeature([], tf.string), \n#     'image_name': tf.io.FixedLenFeature([], tf.string), \n#     'target': tf.io.FixedLenFeature([], tf.int64),\n# }\n\n# def resize_image(img_arr):\n#     new_img = Image.fromarray(img_arr).resize(size=(224, 224))\n#     ret_arr = np.array(new_img)\n#     return ret_arr \n\n# def parse_img_func(example):\n#     return tf.io.parse_single_example(example, image_desc)\n\n# def transform_rec(tfrec):\n#     dataset = tf.data.TFRecordDataset(tfrec)\n# #     print(sys.getsizeof(dataset))\n#     parsed_set = dataset.map(parse_img_func)\n#     img_arrays = [np.array(Image.open(io.BytesIO(i['image'].numpy()))) for i in parsed_set]\n#     img_arrays = np.array(list(map(resize_image, img_arrays)))\n#     img_names = [str(i['image_name'].numpy())[2:-1] for i in parsed_set]\n#     targets = [i['target'].numpy() for i in parsed_set]\n#     return img_arrays, np.array(img_names), np.array(targets)\n\n# start = time.time()\n# img_arrays, img_names, targets = [], [], []\n# for i in tqdm(range(len(tf_rec_files[:2]))):\n#     imgs, names, vals = transform_rec(tfrec_loc+tf_rec_files[i])\n#     img_arrays.append(imgs)\n#     img_names.append(names)\n#     targets.append(vals)\n#     next_arrays, next_names, next_targets = transform_rec(tfrec_loc+'train01-2071.tfrec') \n#     third_arrays, third_names, third_targets = transform_rec(tfrec_loc+'train02-2071.tfrec')\n# imgs, names, vals = transform_rec(tfrec_loc+tf_rec_files[0])\n\n# p = names.argsort()\n# names = names[p]\n# imgs = imgs[p]\n# vals = vals[p]\n\n# train_w_imgs = train_data[train_data['image_name'].isin(names)]\n# train_w_imgs = train_w_imgs.sort_values(by=['image_name'])\n# imgs = tf.cast(imgs, tf.float32)\n# targets = np.c_[vals]\n\n# end = time.time()\n# print(end-start)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# pp_pipeline = Pipeline([\n#     ('Imputer', SimpleImputer(missing_values=np.nan, strategy='mean')), \n#     ('Scaler', preprocessing.StandardScaler())\n# ])\n# unqs = train_w_imgs['patient_id'].unique()\n\n# count_dict = {'patient_id': [], 'location': [], 'count': []}\n# for patient in unqs:\n#     counts = train_w_imgs[train_w_imgs['patient_id'] == patient]['anatom_site_general_challenge'].value_counts()\n#     for i in range(len(counts.index)):\n#         count_dict['patient_id'].append(patient)\n#         count_dict['location'].append(counts.index[i])\n#         count_dict['count'].append(counts.values[i])\n# loc_counter = pd.DataFrame(count_dict)\n# del count_dict\n# gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# full_df = train_w_imgs.merge(loc_counter, left_on=['patient_id', 'anatom_site_general_challenge'], \\\n#                            right_on=['patient_id', 'location'], how='left')\n# del train_w_imgs\n# del loc_counter\n# del full_df['anatom_site_general_challenge']\n\n# cols = list(full_df.columns)\n# cols.remove('diagnosis')\n# cols.remove('benign_malignant')\n# cols.remove('target')\n# cols.remove('patient_id')\n# cols.remove('image_name')\n\n# data_x = full_df[cols]\n# data_x = pd.get_dummies(data_x, columns=['sex', 'location'])\n# data_x = np.c_[data_x]\n# data_y = np.c_[full_df['target']]\n\n# x_train = data_x[:1450]\n# x_test = data_x[1451:]\n\nname_train = names[:1450]\nname_test = names[1451:]\ntrain_imgs = tf.cast(imgs[:1450], tf.float32)\ntest_imgs = tf.cast(imgs[1451:], tf.float32)\ntrain_vals = vals[:1450]\ntest_vals = vals[1451:]\n\nprint(type(name_train))\nprint(type(train_imgs))\nprint(type(test_imgs))\nprint(type(train_vals))\n\ndel names\ndel imgs\ndel vals \ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# x_train_pp = pp_pipeline.fit_transform(x_train)\n# x_test_pp = pp_pipeline.transform(x_test)\n\n# del x_train\n# del x_test\n\n# gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# detect and init the TPU\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver()\ntf.config.experimental_connect_to_cluster(tpu)\ntf.tpu.experimental.initialize_tpu_system(tpu)\n\n# instantiate a distribution strategy\ntpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\n# instantiating the model in the strategy scope creates the model on the TPU\nwith tpu_strategy.scope():\n    print(\"Starting up\")\n    pic_mod = models.Sequential([\n        layers.Conv2D(96, (3,3), activation='relu', input_shape=(224, 224, 3)), \n        layers.MaxPooling2D((3,3)), \n        layers.Conv2D(256, (3,3), activation='relu'),\n        layers.MaxPooling2D((3,3)), \n        layers.Conv2D(384, (3,3), activation='relu'),\n    #     layers.Conv2D(384, (3,3), activation='relu'),\n        layers.Conv2D(256, (3,3), activation='relu'),\n        layers.Flatten(), \n        layers.Dense(128, activation='relu'), \n        layers.Dense(1, activation='sigmoid')\n    ])\n    sgd = optimizers.Adam(lr=0.5)\n    pic_mod.compile(optimizer=sgd, loss='binary_crossentropy', \n                    metrics=[tf.keras.metrics.AUC()])\n    print(\"Ready to train\")\n    time.sleep(4)\nhist = pic_mod.fit(train_imgs, train_vals, batch_size=218, epochs=10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# detect and init the TPU\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n# tf.config.experimental_connect_to_cluster(tpu)\n# tf.tpu.experimental.initialize_tpu_system(tpu)\n\n# # instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\n# time.sleep(1)\n# # instantiating the model in the strategy scope creates the model on the TPU\n# with tpu_strategy.scope():\n    \n#     # Input for the data from the csv \n# #     csv_input = Input(shape=10)\n#     # Input for the Image data \n#     img_input = Input(shape=(224, 224, 3))\n    \n#     # Hidden layers for the csv input \n# #     csv_1 = layers.Dense(64, activation='relu')(csv_input)\n# #     csv_2 = layers.Dense(128, activation='relu')(csv_1)\n# #     csv_3 = layers.Dense(128, activation='relu')(csv_2)\n    \n#     # Convolutional and Pooling hidden layers for the image input \n#     cnn_1 = layers.Conv2D(25, (3,3), activation='relu')(img_input)\n#     pool1 = layers.MaxPooling2D((3,3))(cnn_1)\n#     cnn2 = layers.Conv2D(45, (3,3), activation='relu')(pool1)\n#     pool2 = layers.MaxPooling2D((3,3))(cnn2)\n# #     cnn3 = layers.Conv2D(100, (3,3), activation='relu')(pool2)\n#     flat = layers.Flatten()(pool2)\n#     connected = layers.Dense(45, activation='relu')(flat)\n    \n#     # Concatenate the output from both the csv and image hidden layers \n# #     merger = layers.concatenate([csv_3, connected])\n    \n#     # Accepted the merged together vectors and output \n#     output = layers.Dense(1, activation='sigmoid')(connected)\n    \n#     # Call and compile the model \n# #     mod = Model(inputs=[csv_input, img_input], outputs=output)\n#     mod = Model(inputs=img_input, outputs=output)\n#     sgd = optimizers.Adam(lr=0.005)\n#     mod.compile(optimizer=sgd, loss='binary_crossentropy', metrics=[tf.keras.metrics.AUC()])\n# print(\"Finished\")\n# time.sleep(1)\n# # mod.fit(train_imgs, train_vals, batch_size=145, epochs=10)\n# # mod([x_train_pp[0], train_imgs[0]])\n# mod(train_imgs[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# mod.fit(train_imgs, train_vals, batch_size=218, epochs=10)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}