{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom glob import glob\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport warnings\nwarnings.filterwarnings('ignore')\nprint(os.listdir(\"../input\"))\nimport sys\nimport cv2\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_files       \nfrom keras.utils import np_utils\nfrom sklearn.utils import shuffle\nfrom sklearn.metrics import log_loss\n\nfrom keras.models import Sequential, Model\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, BatchNormalization, GlobalAveragePooling2D\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.preprocessing import image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!git clone https://github.com/recursionpharma/rxrx1-utils\nprint ('rxrx1-utils cloned!')\n!ls\nsys.path.append('rxrx1-utils')\nimport rxrx.io as rio","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"整合数据"},{"metadata":{"trusted":true},"cell_type":"code","source":"md = rio.combine_metadata()\nmd.head(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = md[md['dataset'] == 'train']\ntest_df = md[md['dataset'] == 'test']\ntrain_df.shape, test_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.head(4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.index=list(range(0,len(train_df),1))\ntrain_df.head(4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"experiment =train_df['experiment']\nplate =train_df['plate']\nsirna =train_df['sirna']\nwell =train_df['well']\nsite =train_df['site']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"i=8\ny = rio.load_site_as_rgb('train',experiment[i], plate[i], well[i], site[i])\nplt.figure(figsize=(8, 8))\nplt.axis('off')\nplt.imshow(y)\nprint(y.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"NUMBER_CLASSES = len(train_df['sirna'].unique())\nprint(NUMBER_CLASSES)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_data(num):\n    data_images = [] \n    data_labels = []\n    for i in range(num):                                                                                                                                                                                                                                                                                                                                                                                                                                                                                                 \n        data_images.append(i)           \n        data_labels.append(int(sirna[i]))     \n        if(i%500==0):\n            print(i)\n    return data_images,  data_labels ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data_images,data_labels=load_data(len(train_df))\n#len(dataset)\nprint(len(data_labels))\nprint(len(data_images))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_train, x_val, y_train, y_val = train_test_split(data_images, data_labels, test_size=0.2, random_state=0)\nprint(len(x_train))\nprint(len(y_train))\nprint(len(x_val))\nprint(len(y_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications.inception_v3 import InceptionV3\nbase_model=InceptionV3(\n            include_top=False, \n            weights='imagenet', \n            input_tensor=None, \n            input_shape=None, \n            pooling=None, \n            classes=NUMBER_CLASSES)\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n\n# 添加一个全连接层\nx = Dense(2048, activation='relu')(x)\n# 添加一个分类器\npredictions = Dense(NUMBER_CLASSES, activation='softmax')(x)\n\n# 构建我们需要训练的完整模型\nmodel = Model(inputs=base_model.input, outputs=predictions)\nfor layer in base_model.layers:\n    layer.trainable = False","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()\n\nmodel.compile(optimizer='rmsprop', loss='categorical_crossentropy',metrics=['acc'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from random import randint\ndef generator_data(datas,labels,batch_size):\n    num_batch = len(datas)//batch_size\n    while True:\n        imgs = []\n        i= randint(0,num_batch)\n        train_datas = datas[batch_size*i:batch_size*(i+1)]\n        train_lables = labels[batch_size*i:batch_size*(i+1)]\n        train_lables =np_utils.to_categorical(train_lables,NUMBER_CLASSES ) \n        for temp in train_datas:\n            img = rio.load_site_as_rgb('train',experiment[temp], plate[temp], well[temp], site[temp])\n            imgs.append(img)                        \n        imgs = np.array(imgs, dtype=np.uint8).reshape(-1,512,512,3)\n        yield (imgs,train_lables)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"batch_size=10\nhistory= model.fit_generator(\n                         generator_data(x_train, y_train, batch_size),\n                         #steps_per_epoch =len(x_train)//batch_size,\n                         steps_per_epoch =100,\n                         epochs = 50, \n                         verbose = 1,#日志显示模式\n                         validation_data =generator_data(x_val, y_val,batch_size),\n                         #validation_steps =len(x_val)//batch_size,\n                         validation_steps =20,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nhistory_dict = history.history \nloss_values = history_dict['loss'] \nval_loss_values = history_dict['val_loss'] \n \nepochs = range(1, len(loss_values) + 1) \n \nplt.plot(epochs, loss_values, 'bo', label='Training loss')   \nplt.plot(epochs, val_loss_values, 'b', label='Validation loss')   \nplt.title('Training and validation loss') \nplt.xlabel('Epochs') \nplt.ylabel('Loss') \nplt.legend() \n \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"acc = history_dict['acc']  \nval_acc = history_dict['val_acc'] \n\nplt.plot(epochs, acc, 'bo', label='Training acc') \nplt.plot(epochs, val_acc, 'b', label='Validation acc') \nplt.title('Training and validation accuracy') \nplt.xlabel('Epochs') \nplt.ylabel('Accuracy') \nplt.legend() \n \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save_weights(\"weights.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df.index=list(range(0,len(test_df),1))\ntest_df.head(4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_experiment =test_df['experiment']\ntest_plate=test_df['plate']\ntest_well=test_df['well']\ntest_site=test_df['site']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def prediction_test(num,model):\n    preds =[]\n    for i in range(num):\n        img = rio.load_site_as_rgb('test',test_experiment[i], test_plate[i], test_well[i], test_site[i])\n        x = np.array(img, dtype=np.uint8).reshape(-1,512,512,3)\n        output = model.predict(x) \n        idx = np.argmax(output)\n        preds.append(idx)\n        if(i%1000==0):\n            print(i)\n    return data_images","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = prediction(len(test_images),model)\nprint(len(preds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = pd.read_csv('../input/' + '/test.csv')\nsubmission.head(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission['sirna'] = preds\nsubmission.head(6)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False, columns=['id_code','sirna'])","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}