{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        break\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport seaborn\nimport cv2\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import tensorflow as tf\nimport keras\nfrom keras.layers import Conv2D,MaxPooling2D,Dense,Flatten,BatchNormalization\nfrom keras.models import Sequential,Model","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Count Values ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df.target.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.diagnosis.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.sex.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.anatom_site_general_challenge.value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_img=10\nreshape_size = 512\nchannel = 3","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df = df.head(sample_img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(new_df.head())\nprint(new_df.tail())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n=0\nfor i in range(df.shape[0]):\n    if(df['target'][i]==1):\n        n+=1\n        x =df.iloc[i,:]\n        x = pd.DataFrame(x.values.reshape(1,-1),columns = x.index)\n        new_df=pd.concat([new_df,x],axis=0)\n    if(n==sample_img):\n        break\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df.dropna(axis=0,inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def labelfullpath(df,train=True):\n    base_path=\"../input/siim-isic-melanoma-classification/jpeg\"\n    if(train==True):\n        base_path = os.path.join(base_path,\"train\")\n    else:\n        base_path = os.path.join(base_path,\"test\")\n    fullpath = [os.path.join(base_path,img+\".jpg\") for img in df.image_name]\n    df['fullpath'] = fullpath\n    return df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df = labelfullpath(new_df)\nnew_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df['gender'] = [float(new_df['sex'].values[i]=='female') for i in range(sample_img*2)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dict_anatom = {'oral/genital':0,'palms/soles':0.20,'head/neck':0.40,'upper extremity':0.60,'lower extremity':0.80,'torso':1.0}\nnew_df['anatom_site'] = [dict_anatom[new_df['anatom_site_general_challenge'].values[i]] for i in range(sample_img*2)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df['age'] = [(new_df['age_approx'].values[i])/100.0 for i in range(sample_img*2)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = new_df[['gender','anatom_site','age']].values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Plot some Images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"new_df.fullpath.values[0]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Benign (0) images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,10))\nnum = 1\nfor i in range(5):\n    plt.subplot(1,5,num)\n    im = plt.imread(new_df.fullpath.values[i])\n    plt.imshow(im)\n    plt.title(im.shape)\n    plt.xlabel(new_df.sex.values[i]+\" \"+str(new_df.age_approx.values[i])+\"\\n\"+new_df.anatom_site_general_challenge.values[i])\n    num+=1","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# malignant (1) images","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize = (20,10))\nnum = 1\nfor i in range(5):\n    plt.subplot(1,5,num)\n    im = plt.imread(new_df.fullpath.values[i+5])\n    plt.imshow(im)\n    plt.title(im.shape)\n    plt.xlabel(new_df.sex.values[i+5]+\" \"+str(new_df.age_approx.values[i+5])+\"\\n\"+new_df.anatom_site_general_challenge.values[i+5])\n    num+=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def preprocessing_images(imglist,channel=1):\n    image_arr =[] \n    for img in imglist:\n        if(channel==1):\n            i = cv2.imread(img,cv2.IMREAD_GRAYSCALE)\n        else:\n            i = cv2.imread(img)\n        i = cv2.resize(i,(reshape_size,reshape_size))\n        i = i/255.0\n        image_arr.append(i)\n    return np.array(image_arr)    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images = preprocessing_images(new_df.fullpath.values,channel=channel)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"images.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shuffle_index = [i for i in range(0,2*sample_img)]\nnp.random.shuffle(shuffle_index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shuffle_images = images[shuffle_index]\nshuffle_labels = new_df.target.values[shuffle_index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if(channel==1):    \n    fit_images = np.expand_dims(shuffle_images,axis=3)\nelse:\n    fit_images = shuffle_images\nonehot_labels = np.array([np.eye(2)[i] for i in shuffle_labels])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(fit_images.shape)\nprint(onehot_labels.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.imshow(shuffle_images[1],cmap = 'hot')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def simple_model():\n    keras.backend.clear_session()\n    \n    ip1 = keras.layers.Input(shape = (reshape_size,reshape_size,channel))\n    ip2 = keras.layers.Input(shape = (3,))\n    vgg = keras.applications.VGG19(input_shape=(reshape_size,reshape_size,channel),include_top=False,weights = 'imagenet')(ip1)\n    vgg.trainable = False\n    flat = Flatten()(vgg)\n    Dense1 =  Dense(525,activation='relu')(flat)\n    Dense2 = Dense(525,activation='relu')(Dense1)\n    Dense3 = Dense(50,activation='relu')(ip2)\n    Dense4 = Dense(50,activation='relu')(Dense3)\n    concatelayer = keras.layers.Concatenate(axis=1)([Dense2,Dense4])\n    DenseL1 = Dense(228,activation='relu')(concatelayer)\n    output1 = Dense(2,activation='softmax')(DenseL1)\n    mainmodel = Model(inputs = [ip1,ip2],outputs = output1)                 \n    mainmodel.compile('adam','categorical_crossentropy',metrics = ['accuracy'])\n    print(mainmodel.summary())\n    print(\"input shape \",mainmodel.input_shape)\n    print(\"output shape \",mainmodel.output_shape)\n    return mainmodel","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = simple_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"hist = model.fit([fit_images,x],onehot_labels,epochs=10,batch_size=16,validation_split=0.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(10,7))\nplt.subplot(1,2,1)\nplt.plot(hist.history['accuracy'],label='accuracy')\nplt.plot(hist.history['loss'],label='loss')\nplt.legend()\nplt.title(\"training set\")\nplt.grid()\nplt.subplot(1,2,2)\nplt.plot(hist.history['val_accuracy'],label='val_accuracy')\nplt.plot(hist.history['val_loss'],label='val_loss')\nplt.legend()\nplt.title(\"validation set\")\nplt.grid()\nplt.ylim((0,4))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}