{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\nimport tensorflow as tf\nimport os\nfrom glob import glob\nfrom random import shuffle\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.layers import Convolution1D, concatenate, SpatialDropout1D, GlobalMaxPool1D, GlobalAvgPool1D, Embedding, \\\n    Conv2D, SeparableConv1D, Add, BatchNormalization, Activation, GlobalAveragePooling2D, LeakyReLU, Flatten\nfrom keras.layers import Dense, Input, Dropout, MaxPooling2D, Concatenate, GlobalMaxPooling2D, GlobalAveragePooling2D, \\\n    Lambda, Multiply, LSTM, Bidirectional, PReLU, MaxPooling1D\nfrom keras.layers.pooling import _GlobalPooling1D\nfrom keras.losses import mae, sparse_categorical_crossentropy, binary_crossentropy\nfrom keras.models import Model\nfrom keras.applications.nasnet import NASNetMobile, NASNetLarge, preprocess_input\nfrom keras.optimizers import Adam, RMSprop\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\nfrom imgaug import augmenters as iaa\nimport imgaug as ia\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-17T20:44:47.878459Z","iopub.execute_input":"2023-05-17T20:44:47.878850Z","iopub.status.idle":"2023-05-17T20:44:49.284534Z","shell.execute_reply.started":"2023-05-17T20:44:47.878778Z","shell.execute_reply":"2023-05-17T20:44:49.283665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/train_labels.csv\")\nid_label_map = {k:v for k,v in zip(df_train.id.values, df_train.label.values)}\ndf_train.head()","metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","execution":{"iopub.status.busy":"2023-05-07T11:38:33.102436Z","iopub.execute_input":"2023-05-07T11:38:33.102803Z","iopub.status.idle":"2023-05-07T11:38:33.642403Z","shell.execute_reply.started":"2023-05-07T11:38:33.102743Z","shell.execute_reply":"2023-05-07T11:38:33.641654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_id_from_file_path(file_path):\n    return file_path.split(os.path.sep)[-1].replace('.tif', '')","metadata":{"_uuid":"4c1169ec9a84704cff822b6e8ba90729d0ee383e","execution":{"iopub.status.busy":"2023-05-07T11:38:38.779001Z","iopub.execute_input":"2023-05-07T11:38:38.779326Z","iopub.status.idle":"2023-05-07T11:38:38.783832Z","shell.execute_reply.started":"2023-05-07T11:38:38.779265Z","shell.execute_reply":"2023-05-07T11:38:38.782895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labeled_files = glob('../input/train/*.tif')\ntest_files = glob('../input/test/*.tif')","metadata":{"_uuid":"4839c33e47619dfaf0f63d29bcf01ba50a96dfec","execution":{"iopub.status.busy":"2023-05-07T11:38:43.05827Z","iopub.execute_input":"2023-05-07T11:38:43.058599Z","iopub.status.idle":"2023-05-07T11:38:51.596308Z","shell.execute_reply.started":"2023-05-07T11:38:43.05854Z","shell.execute_reply":"2023-05-07T11:38:51.595395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"labeled_files size :\", len(labeled_files))\nprint(\"test_files size :\", len(test_files))","metadata":{"_uuid":"4d3168a806b716023cef1d798b3ede847f7546ee","execution":{"iopub.status.busy":"2023-05-07T11:39:01.985795Z","iopub.execute_input":"2023-05-07T11:39:01.986104Z","iopub.status.idle":"2023-05-07T11:39:01.99833Z","shell.execute_reply.started":"2023-05-07T11:39:01.986045Z","shell.execute_reply":"2023-05-07T11:39:01.997112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train, val = train_test_split(labeled_files, test_size=0.1, random_state=101010)","metadata":{"_uuid":"af19d597affefcbb1a14e5dd0d591bb3294efb61","execution":{"iopub.status.busy":"2023-05-07T11:39:23.708926Z","iopub.execute_input":"2023-05-07T11:39:23.709249Z","iopub.status.idle":"2023-05-07T11:39:23.772135Z","shell.execute_reply.started":"2023-05-07T11:39:23.709177Z","shell.execute_reply":"2023-05-07T11:39:23.771195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def chunker(seq, size):\n    return (seq[pos:pos + size] for pos in range(0, len(seq), size))\ndef get_seq():\n    sometimes = lambda aug: iaa.Sometimes(0.5, aug)\n    seq = iaa.Sequential(\n        [\n            # apply the following augmenters to most images\n            iaa.Fliplr(0.5), # horizontally flip 50% of all images\n            iaa.Flipud(0.2), # vertically flip 20% of all images\n            sometimes(iaa.Affine(\n                scale={\"x\": (0.9, 1.1), \"y\": (0.9, 1.1)}, # scale images to 80-120% of their size, individually per axis\n                translate_percent={\"x\": (-0.1, 0.1), \"y\": (-0.1, 0.1)}, # translate by -20 to +20 percent (per axis)\n                rotate=(-10, 10), # rotate by -45 to +45 degrees\n                shear=(-5, 5), # shear by -16 to +16 degrees\n                order=[0, 1], # use nearest neighbour or bilinear interpolation (fast)\n                cval=(0, 255), # if mode is constant, use a cval between 0 and 255\n                mode=ia.ALL # use any of scikit-image's warping modes (see 2nd image from the top for examples)\n            )),\n            # execute 0 to 5 of the following (less important) augmenters per image\n            # don't execute all of them, as that would often be way too strong\n            iaa.SomeOf((0, 5),\n                [\n                    sometimes(iaa.Superpixels(p_replace=(0, 1.0), n_segments=(20, 200))), # convert images into their superpixel representation\n                    iaa.OneOf([\n                        iaa.GaussianBlur((0, 1.0)), # blur images with a sigma between 0 and 3.0\n                        iaa.AverageBlur(k=(3, 5)), # blur image using local means with kernel sizes between 2 and 7\n                        iaa.MedianBlur(k=(3, 5)), # blur image using local medians with kernel sizes between 2 and 7\n                    ]),\n                    iaa.Sharpen(alpha=(0, 1.0), lightness=(0.9, 1.1)), # sharpen images\n                    iaa.Emboss(alpha=(0, 1.0), strength=(0, 2.0)), # emboss images\n                    # search either for all edges or for directed edges,\n                    # blend the result with the original image using a blobby mask\n                    iaa.SimplexNoiseAlpha(iaa.OneOf([\n                        iaa.EdgeDetect(alpha=(0.5, 1.0)),\n                        iaa.DirectedEdgeDetect(alpha=(0.5, 1.0), direction=(0.0, 1.0)),\n                    ])),\n                    iaa.AdditiveGaussianNoise(loc=0, scale=(0.0, 0.01*255), per_channel=0.5), # add gaussian noise to images\n                    iaa.OneOf([\n                        iaa.Dropout((0.01, 0.05), per_channel=0.5), # randomly remove up to 10% of the pixels\n                        iaa.CoarseDropout((0.01, 0.03), size_percent=(0.01, 0.02), per_channel=0.2),\n                    ]),\n                    iaa.Invert(0.01, per_channel=True), # invert color channels\n                    iaa.Add((-2, 2), per_channel=0.5), # change brightness of images (by -10 to 10 of original value)\n                    iaa.AddToHueAndSaturation((-1, 1)), # change hue and saturation\n                    # either change the brightness of the whole image (sometimes\n                    # per channel) or change the brightness of subareas\n                    iaa.OneOf([\n                        iaa.Multiply((0.9, 1.1), per_channel=0.5),\n                        iaa.FrequencyNoiseAlpha(\n                            exponent=(-1, 0),\n                            first=iaa.Multiply((0.9, 1.1), per_channel=True),\n                            second=iaa.ContrastNormalization((0.9, 1.1))\n                        )\n                    ]),\n                    sometimes(iaa.ElasticTransformation(alpha=(0.5, 3.5), sigma=0.25)), # move pixels locally around (with random strengths)\n                    sometimes(iaa.PiecewiseAffine(scale=(0.01, 0.05))), # sometimes move parts of the image around\n                    sometimes(iaa.PerspectiveTransform(scale=(0.01, 0.1)))\n                ],\n                random_order=True\n            )\n        ],\n        random_order=True\n    )\n    return seq\n\ndef data_gen(list_files, id_label_map, batch_size, augment=False):\n    seq = get_seq()\n    while True:\n        shuffle(list_files)\n        for batch in chunker(list_files, batch_size):\n            X = [cv2.imread(x) for x in batch]\n            Y = [id_label_map[get_id_from_file_path(x)] for x in batch]\n            if augment:\n                X = seq.augment_images(X)\n            X = [preprocess_input(x) for x in X]\n                \n            yield np.array(X), np.array(Y)\n    ","metadata":{"_uuid":"bd67e144f327114229b410e0af43b71f45ce0d72","execution":{"iopub.status.busy":"2023-05-07T11:39:28.710726Z","iopub.execute_input":"2023-05-07T11:39:28.71109Z","iopub.status.idle":"2023-05-07T11:39:28.739346Z","shell.execute_reply.started":"2023-05-07T11:39:28.711027Z","shell.execute_reply":"2023-05-07T11:39:28.738359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_classif_nasnet():\n    inputs = Input((96, 96, 3))\n    base_model = NASNetMobile(include_top=False, input_shape=(96, 96, 3))#, weights=None\n    x = base_model(inputs)\n    out1 = GlobalMaxPooling2D()(x)\n    out2 = GlobalAveragePooling2D()(x)\n    out3 = Flatten()(x)\n    out = Concatenate(axis=-1)([out1, out2, out3])\n    out = Dropout(0.5)(out)\n    out = Dense(1, activation=\"sigmoid\", name=\"3_\")(out)\n    model = Model(inputs, out)\n    model.compile(optimizer=Adam(0.0001), loss=binary_crossentropy, metrics=['acc'])\n    model.summary()\n\n    return model","metadata":{"_uuid":"ad80bc3b50a313d8fd09637cc40c390e0042c765","execution":{"iopub.status.busy":"2023-05-17T18:06:42.612840Z","iopub.execute_input":"2023-05-17T18:06:42.613168Z","iopub.status.idle":"2023-05-17T18:06:42.621759Z","shell.execute_reply.started":"2023-05-17T18:06:42.613109Z","shell.execute_reply":"2023-05-17T18:06:42.620852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = get_model_classif_nasnet()","metadata":{"_uuid":"0872139964e7b1e7a25b327941b68fc3f90daa46","execution":{"iopub.status.busy":"2023-05-17T18:06:46.552843Z","iopub.execute_input":"2023-05-17T18:06:46.553198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=32\nh5_path = '/kaggle/input/nasnet/model.h5'\ncheckpoint = ModelCheckpoint(h5_path, monitor='val_acc', verbose=1, save_best_only=True, mode='max')\n\nhistory = model.fit_generator(\n    data_gen(train, id_label_map, batch_size, augment=True),\n    validation_data=data_gen(val, id_label_map, batch_size),\n    epochs=2, verbose=1,\n    callbacks=[checkpoint],\n    steps_per_epoch=len(train) // batch_size,\n    validation_steps=len(val) // batch_size)\nbatch_size=64\nhistory = model.fit_generator(\n    data_gen(train, id_label_map, batch_size, augment=True),\n    validation_data=data_gen(val, id_label_map, batch_size),\n    epochs=6, verbose=1,\n    callbacks=[checkpoint],\n    steps_per_epoch=len(train) // batch_size,\n    validation_steps=len(val) // batch_size)\n\nmodel.load_weights(h5_path)","metadata":{"_uuid":"db0b06f165a011723aef4266bf1be50b977268c5","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.load_weights('/kaggle/input/nasnet/model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-05-17T18:05:59.966945Z","iopub.execute_input":"2023-05-17T18:05:59.967292Z","iopub.status.idle":"2023-05-17T18:05:59.985254Z","shell.execute_reply.started":"2023-05-17T18:05:59.967233Z","shell.execute_reply":"2023-05-17T18:05:59.983934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('/kaggle/working/model.h5')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n#model = load_model(\"/kaggle/input/nasnet/model.h5\")\npath=\"/kaggle/input/nasnet/model.h5\"\nmodel=path.load()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T18:04:06.224741Z","iopub.execute_input":"2023-05-17T18:04:06.225094Z","iopub.status.idle":"2023-05-17T18:04:06.244872Z","shell.execute_reply.started":"2023-05-17T18:04:06.225028Z","shell.execute_reply":"2023-05-17T18:04:06.243574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/train_labels.csv\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nids = []","metadata":{"_uuid":"d399c51b099a414ef83e9b45a8bafab7206b1c24","execution":{"iopub.status.busy":"2023-05-07T18:00:04.629304Z","iopub.execute_input":"2023-05-07T18:00:04.629909Z","iopub.status.idle":"2023-05-07T18:00:04.635708Z","shell.execute_reply.started":"2023-05-07T18:00:04.629663Z","shell.execute_reply":"2023-05-07T18:00:04.634716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for batch in chunker(test_files, batch_size):\n    X = [preprocess_input(cv2.imread(x)) for x in batch]\n    ids_batch = [get_id_from_file_path(x) for x in batch]\n    X = np.array(X)\n    preds_batch = ((model.predict(X).ravel()*model.predict(X[:, ::-1, :, :]).ravel()*model.predict(X[:, ::-1, ::-1, :]).ravel()*model.predict(X[:, :, ::-1, :]).ravel())**0.25).tolist()\n    preds += preds_batch\n    ids += ids_batch","metadata":{"_uuid":"787cec513417ebb2c703d48b0ba97e1c4344c8d7","execution":{"iopub.status.busy":"2023-05-07T18:00:14.036874Z","iopub.execute_input":"2023-05-07T18:00:14.037178Z","iopub.status.idle":"2023-05-07T18:16:05.681425Z","shell.execute_reply.started":"2023-05-07T18:00:14.037123Z","shell.execute_reply":"2023-05-07T18:16:05.680274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({'id':ids, 'label':preds})\ndf.to_csv(\"baseline_nasnet.csv\", index=False)\ndf.head()","metadata":{"_uuid":"1b531bfe1d28a7bae83b74d9f857ae0f7029fdd2","execution":{"iopub.status.busy":"2023-05-07T18:28:56.60696Z","iopub.execute_input":"2023-05-07T18:28:56.60729Z","iopub.status.idle":"2023-05-07T18:28:57.020407Z","shell.execute_reply.started":"2023-05-07T18:28:56.607232Z","shell.execute_reply":"2023-05-07T18:28:57.019541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save(working + \".csv\", df)\nworking='/kaggle/working/output_nasnet'","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/nasnet/baseline_nasnet (1).csv\")\nid_label_map = {k:v for k,v in zip(df.id.values, df.label.values)}\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-17T20:45:02.218740Z","iopub.execute_input":"2023-05-17T20:45:02.219073Z","iopub.status.idle":"2023-05-17T20:45:02.395300Z","shell.execute_reply.started":"2023-05-17T20:45:02.219017Z","shell.execute_reply":"2023-05-17T20:45:02.394460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfig = plt.figure(figsize=(25, 20))\npath2data = \"/kaggle/input/histopathologic-cancer-detection/test\"\ntest_imgs = os.listdir(path2data)\nfor idx, img in enumerate(np.random.choice(test_imgs, 5)):\n    ax = fig.add_subplot(3, 20//3, idx+1)\n    im = Image.open(path2data + \"/\" + img)\n    plt.imshow(im)\n    lab = df.loc[df[\"id\"] == img.split('.')[0], 'label'].values[0]\n    if lab>=0.5:\n        labs=1\n    else:\n        labs=0\n    lab=format(lab, '.3f')\n    ax.set_title(f'Label: {labs} \\n Prediction: {lab}')\n    ax.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-05-17T21:19:11.674349Z","iopub.execute_input":"2023-05-17T21:19:11.674690Z","iopub.status.idle":"2023-05-17T21:19:12.713511Z","shell.execute_reply.started":"2023-05-17T21:19:11.674631Z","shell.execute_reply":"2023-05-17T21:19:12.712638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nfig = plt.figure(figsize=(3, 4)) #resize براحتك :D\npath2data = \"/kaggle/input/histopathologic-cancer-detection/test\"\nimg='bd953a3b1db1f7041ee95ff482594c4f46c73ed0.tif' #enter the name here\nim = Image.open(path2data + \"/\" + img)\nplt.imshow(im)\nlab = df.loc[df[\"id\"] == img.split('.')[0], 'label'].values[0]\nif lab>=0.5:\n    labs=1\nelse:\n    labs=0\nlab=format(lab, '.3f')\nplt.title(f'Label: {labs} \\n Prediction: {lab}')\nplt.axis('off')\n","metadata":{"execution":{"iopub.status.busy":"2023-05-17T21:22:02.240686Z","iopub.execute_input":"2023-05-17T21:22:02.241026Z","iopub.status.idle":"2023-05-17T21:22:02.371988Z","shell.execute_reply.started":"2023-05-17T21:22:02.240966Z","shell.execute_reply":"2023-05-17T21:22:02.368728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### * * * * * * * * * * ### GOOD LUCK  ### * * * * * * * * * * * * * ","metadata":{}}]}