{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\n%matplotlib inline\nprint(os.listdir(\"../input\"))\nimport sys\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_files       \nfrom keras.utils import np_utils\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import log_loss\n\n\nfrom keras.models import Sequential, Model\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, GlobalAveragePooling2D\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.preprocessing import image\nimport warnings\nwarnings.filterwarnings('ignore')\n# Any results you write to the current directory are saved as output.\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!git clone https://github.com/recursionpharma/rxrx1-utils\nprint ('rxrx1-utils cloned!')\n!ls\nsys.path.append('rxrx1-utils')\nimport rxrx.io as rio\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nmd = rio.combine_metadata()\nmd.head(6)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = md[md['dataset'] == 'train']\ntest_df = md[md['dataset'] == 'test']\ntrain_df.shape, test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.index=list(range(0,len(train_df),1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"experiment =train_df['experiment']\nplate =train_df['plate']\nsirna =train_df['sirna']\nwell =train_df['well']\nsite =train_df['site']\n\n\nNUMBER_CLASSES = len(train_df['sirna'].unique())\nprint(NUMBER_CLASSES)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_data(num):\n    data_images = [] \n    data_labels = []\n    for i in range(num):  \n        data_images.append(i)   \n        data_labels.append(sirna[i]) \n        if(i%500==0):\n            print(i)\n    return data_images,data_labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_images,data_labels = load_data(len(train_df))\nprint(len(data_labels))\nprint(len(data_images))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"819e79b1-3fe7-46af-aa9e-e6e90f4e297c","_cell_guid":"c8c9601a-8d07-4956-b223-402cc67b293c","trusted":true},"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(data_images, data_labels, test_size=0.2, random_state=0)\nprint(len(x_train))\nprint(len(y_train))\nprint(len(x_val))\nprint(len(y_val))\n\nfrom keras.applications.inception_v3 import InceptionV3\n\n\n#--coding:utf-8--\n\n#获得模型信息的代码\n\nfrom keras.applications.densenet import DenseNet201,preprocess_input\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.models import Model\n\n#base_model = DenseNet(weights='imagenet', include_top=False)\nbase_model = DenseNet201(include_top=False)\nx = base_model.output\nx = GlobalAveragePooling2D()(x) \n\npredictions = Dense(5, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\nmodel.summary()\n\n#print('the number of layers in this model:'+str(len(model.layers)))\n# 添加一个全连接层\n#x = Dense(2048, activation='relu')(x)\n# 添加一个分类器\n#predictions = Dense(NUMBER_CLASSES, activation='softmax')(x)\n# 构建我们需要训练的完整模型\n#model.compile(optimizer='rmsprop', loss='categorical_crossentropy',metrics=['acc'])\ndef add_new_last_layer(base_model, nb_classes):  \n    \"\"\"  添加最后的层  输入  base_model和分类数量  输出  新的keras的model  \"\"\"  \n    x = base_model.output  \n    x = GlobalAveragePooling2D()(x)  \n    predictions = Dense(nb_classes, activation='softmax')(x) #new softmax layer  \n    model = Model(input=base_model.input, output=predictions)  \n    return model \n\n\nmodel = DenseNet201(include_top=False)\nmodel = add_new_last_layer(model, NUMBER_CLASSES)\nmodel.compile(optimizer='rmsprop', loss='categorical_crossentropy', metrics=['acc'])\n\n\nfrom random import randint\ndef generator_data(datas,labels,batch_size):\n    num_batch = len(datas)//batch_size\n    while True:\n        imgs = []\n        i= randint(0,num_batch)\n        train_datas = datas[batch_size*i:batch_size*(i+1)]\n        train_lables = labels[batch_size*i:batch_size*(i+1)]\n        train_lables =np_utils.to_categorical(train_lables,NUMBER_CLASSES ) \n        for temp in train_datas:\n            img = rio.load_site_as_rgb('train',experiment[temp], plate[temp], well[temp], site[temp])\n            imgs.append(img)                        \n        imgs = np.array(imgs, dtype=np.uint8).reshape(-1,512,512,3)\n        yield (imgs,train_lables)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size=50\nhistory= model.fit_generator(\n                         generator_data(x_train, y_train, batch_size),\n                         #steps_per_epoch =len(x_train)//batch_size,\n                         steps_per_epoch =100,\n                         epochs = 50, \n                         verbose = 1,#日志显示模式\n                         validation_data =generator_data(x_val, y_val,batch_size),\n                         #validation_steps =len(x_val)//batch_size,\n                         validation_steps =20,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nhistory_dict = history.history \nloss_values = history_dict['loss'] \nval_loss_values = history_dict['val_loss'] \n \nepochs = range(1, len(loss_values) + 1) \n \nplt.plot(epochs, loss_values, 'bo', label='Training loss')   \nplt.plot(epochs, val_loss_values, 'b', label='Validation loss')   \nplt.title('Training and validation loss') \nplt.xlabel('Epochs') \nplt.ylabel('Loss') \nplt.legend() \n \nplt.show()\n\n\n\nacc = history_dict['acc']  \nval_acc = history_dict['val_acc'] \n\nplt.plot(epochs, acc, 'bo', label='Training acc') \nplt.plot(epochs, val_acc, 'b', label='Validation acc') \nplt.title('Training and validation accuracy') \nplt.xlabel('Epochs') \nplt.ylabel('Accuracy') \nplt.legend() \n \nplt.show()\n\n\n\n\nmodel.save_weights(\"weights.h5\")\n\n\ntest_df.index=list(range(0,len(test_df),1))\ntest_df.head(4)\n\n\ntest_experiment =test_df['experiment']\ntest_plate=test_df['plate']\ntest_well=test_df['well']\ntest_site=test_df['site']\n\n\ndef prediction_test(num,model):\n    preds =[]\n    for i in range(num):\n        img = rio.load_site_as_rgb('test',test_experiment[i], test_plate[i], test_well[i], test_site[i])\n        x = np.array(img, dtype=np.uint8).reshape(-1,512,512,3)\n        output = model.predict(x) \n        idx = np.argmax(output)\n        preds.append(idx)\n        if(i%1000==0):\n            print(i)\n    return data_images\n\n\n\npreds = prediction(len(test_df),model)\nprint(len(preds))\n\n\nsubmission = pd.read_csv('../input/' + '/test.csv')\nsubmission.head(6)\n\nsubmission['sirna'] = preds\nsubmission.head(6)\n\nsubmission.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a75f3422-cb5d-4a8a-a13e-6b0fe0cfa427","_cell_guid":"e9006919-f3e0-4cef-9274-77a897152842","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\n%matplotlib inline\nprint(os.listdir(\"../input\"))\nimport sys\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_files       \nfrom keras.utils import np_utils\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import log_loss\n\n\nfrom keras.models import Sequential, Model\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, GlobalAveragePooling2D\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.preprocessing import image\nimport warnings\nwarnings.filterwarnings('ignore')\n# Any results you write to the current directory are saved as output.\n\n\n!git clone https://github.com/recursionpharma/rxrx1-utils\nprint ('rxrx1-utils cloned!')\n!ls\nsys.path.append('rxrx1-utils')\nimport rxrx.io as rio\n\nmd = rio.combine_metadata()\nmd.head(6)\n\n\ntrain_df = md[md['dataset'] == 'train']\ntest_df = md[md['dataset'] == 'test']\ntrain_df.shape, test_df.shape\n\n\nexperiment =train_df['experiment']\nplate =train_df['plate']\nsirna =train_df['sirna']\nwell =train_df['well']\nsite =train_df['site']\n\n\nNUMBER_CLASSES = len(train_df['sirna'].unique())\nprint(NUMBER_CLASSES)\n\n\ndef load_data(num):\n    data_images = [] \n    data_labels = []\n    for i in range(num):                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 \n        data_images.append(i)           \n        data_labels.append(int(sirna[i]))     \n        if(i%500==0):\n            print(i)\n    return data_images,  data_labels \n\n\n\ndata_images,data_labels=load_data(len(train_df))\n#len(dataset)\nprint(len(data_labels))\nprint(len(data_images))\n\n\nx_train, x_val, y_train, y_val = train_test_split(data_images, data_labels, test_size=0.2, random_state=0)\nprint(len(x_train))\nprint(len(y_train))\nprint(len(x_val))\nprint(len(y_val))\n\nfrom keras.applications.inception_v3 import InceptionV3\n\n\n#--coding:utf-8--\n\n#获得模型信息的代码\n\nfrom keras.applications.densenet import DenseNet201,preprocess_input\nfrom keras.layers import Dense, GlobalAveragePooling2D\nfrom keras.models import Model\n\n#base_model = DenseNet(weights='imagenet', include_top=False)\nbase_model = DenseNet201(include_top=False)\nx = base_model.output\nx = GlobalAveragePooling2D()(x) \n\npredictions = Dense(5, activation='softmax')(x)\n\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\nmodel.summary()\n\n#print('the number of layers in this model:'+str(len(model.layers)))\n# 添加一个全连接层\n#x = Dense(2048, activation='relu')(x)\n# 添加一个分类器\n#predictions = Dense(NUMBER_CLASSES, activation='softmax')(x)\n# 构建我们需要训练的完整模型\n#model.compile(optimizer='rmsprop', loss='categorical_crossentropy',metrics=['acc'])\ndef add_new_last_layer(base_model, nb_classes):  \n    \"\"\"  添加最后的层  输入  base_model和分类数量  输出  新的keras的model  \"\"\"  \n    x = base_model.output  \n    x = GlobalAveragePooling2D()(x)  \n    predictions = Dense(nb_classes, activation='softmax')(x) #new softmax layer  \n    model = Model(input=base_model.input, output=predictions)  \n    return model \n\n\nmodel = DenseNet201(include_top=False)\nmodel = add_new_last_layer(model, NUMBER_CLASSES)\nmodel.compile(optimizer='rmsprop', loss='categorical_crossentropy', metrics=['acc'])\n\n\nfrom random import randint\ndef generator_data(datas,labels,batch_size):\n    num_batch = len(datas)//batch_size\n    while True:\n        imgs = []\n        i= randint(0,num_batch)\n        train_datas = datas[batch_size*i:batch_size*(i+1)]\n        train_lables = labels[batch_size*i:batch_size*(i+1)]\n        train_lables =np_utils.to_categorical(train_lables,NUMBER_CLASSES ) \n        for temp in train_datas:\n            img = rio.load_site_as_rgb('train',experiment[temp], plate[temp], well[temp], site[temp])\n            imgs.append(img)                        \n        imgs = np.array(imgs, dtype=np.uint8).reshape(-1,512,512,3)\n        yield (imgs,train_lables)\n\n\n\nbatch_size=100\nhistory= model.fit_generator(\n                         generator_data(x_train, y_train, batch_size),\n                         #steps_per_epoch =len(x_train)//batch_size,\n                         steps_per_epoch =100,\n                         epochs = 50, \n                         verbose = 1,#日志显示模式\n                         validation_data =generator_data(x_val, y_val,batch_size),\n                         #validation_steps =len(x_val)//batch_size,\n                         validation_steps =20,\n)\n\n\n\n\nimport matplotlib.pyplot as plt\nhistory_dict = history.history \nloss_values = history_dict['loss'] \nval_loss_values = history_dict['val_loss'] \n \nepochs = range(1, len(loss_values) + 1) \n \nplt.plot(epochs, loss_values, 'bo', label='Training loss')   \nplt.plot(epochs, val_loss_values, 'b', label='Validation loss')   \nplt.title('Training and validation loss') \nplt.xlabel('Epochs') \nplt.ylabel('Loss') \nplt.legend() \n \nplt.show()\n\n\n\nacc = history_dict['acc']  \nval_acc = history_dict['val_acc'] \n\nplt.plot(epochs, acc, 'bo', label='Training acc') \nplt.plot(epochs, val_acc, 'b', label='Validation acc') \nplt.title('Training and validation accuracy') \nplt.xlabel('Epochs') \nplt.ylabel('Accuracy') \nplt.legend() \n \nplt.show()\n\n\n\n\nmodel.save_weights(\"weights.h5\")\n\n\ntest_df.index=list(range(0,len(test_df),1))\ntest_df.head(4)\n\n\ntest_experiment =test_df['experiment']\ntest_plate=test_df['plate']\ntest_well=test_df['well']\ntest_site=test_df['site']\n\n\ndef prediction_test(num,model):\n    preds =[]\n    for i in range(num):\n        img = rio.load_site_as_rgb('test',test_experiment[i], test_plate[i], test_well[i], test_site[i])\n        x = np.array(img, dtype=np.uint8).reshape(-1,512,512,3)\n        output = model.predict(x) \n        idx = np.argmax(output)\n        preds.append(idx)\n        if(i%1000==0):\n            print(i)\n    return data_images\n\n\n\npreds = prediction(len(test_df),model)\nprint(len(preds))\n\n\nsubmission = pd.read_csv('../input/' + '/test.csv')\nsubmission.head(6)\n\nsubmission['sirna'] = preds\nsubmission.head(6)\n\nsubmission.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","execution_count":null,"outputs":[]}],"metadata":{"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}