{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # 1. Thêm thư viện và các gói hỗ trợ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport cv2\nimport pandas as pd\nimport seaborn as sns\nimport keras\nimport tensorflow_addons as tfa\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.applications import DenseNet169\nfrom keras.layers import Dense,Dropout,Flatten\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom keras.models import Model\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:39.702102Z","iopub.execute_input":"2021-05-25T19:10:39.702465Z","iopub.status.idle":"2021-05-25T19:10:45.184412Z","shell.execute_reply.started":"2021-05-25T19:10:39.702391Z","shell.execute_reply":"2021-05-25T19:10:45.183571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # 2. Đọc các file dữ liệu","metadata":{}},{"cell_type":"code","source":"df=pd.read_csv(\"../input/plant-pathology-2021-fgvc8/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.185806Z","iopub.execute_input":"2021-05-25T19:10:45.186155Z","iopub.status.idle":"2021-05-25T19:10:45.238570Z","shell.execute_reply.started":"2021-05-25T19:10:45.186120Z","shell.execute_reply":"2021-05-25T19:10:45.237620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.240527Z","iopub.execute_input":"2021-05-25T19:10:45.240869Z","iopub.status.idle":"2021-05-25T19:10:45.262775Z","shell.execute_reply.started":"2021-05-25T19:10:45.240832Z","shell.execute_reply":"2021-05-25T19:10:45.261588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Ta có 2 tập dữ liệu dùng để huấn luyện:  \n1. Tập dữ liệu là file .csv chứa 2 cột:  \n- Cột 1 là tên của tất cả các ảnh (**image**)\n- Cột 2 là các loại bệnh tương ứng với mỗi chiếc ảnh bên cột 1 (**labels**)  \n2. Tập dữ liệu là file ảnh chứa tất cả gồm 18632 ảnh màu và tất cả các ảnh đều có tên trong file .csv ở trên.","metadata":{}},{"cell_type":"markdown","source":"> # 3. Tiền xử lý","metadata":{}},{"cell_type":"markdown","source":"### Visualize dữ liệu","metadata":{}},{"cell_type":"code","source":"ax = plt.subplots(figsize=(18, 6))\nsns.set_style(\"whitegrid\")\nsns.countplot(x='labels', data=df);\nplt.xticks(rotation=90);","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.266180Z","iopub.execute_input":"2021-05-25T19:10:45.268633Z","iopub.status.idle":"2021-05-25T19:10:45.661161Z","shell.execute_reply.started":"2021-05-25T19:10:45.268604Z","shell.execute_reply":"2021-05-25T19:10:45.660385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.labels.value_counts().to_frame().style.background_gradient(cmap=\"plasma\")","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.662443Z","iopub.execute_input":"2021-05-25T19:10:45.662772Z","iopub.status.idle":"2021-05-25T19:10:45.772312Z","shell.execute_reply.started":"2021-05-25T19:10:45.662735Z","shell.execute_reply":"2021-05-25T19:10:45.771471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\ncolors = ['gold', 'mediumturquoise', 'darkorange', 'lightgreen']\n\nlabel_counts = df['labels'].value_counts()\nfig = go.Figure(data=[go.Pie(labels=label_counts.index,values=label_counts)])\nfig.update_traces(hoverinfo='label+percent', textinfo='value', textfont_size=20,\n                  marker=dict(colors=colors, line=dict(color='#000000', width=2)))\nfig.update_layout(title='Labels distribution')\nfig.show()\nplt.savefig('labels1.png',transparent=True)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.773813Z","iopub.execute_input":"2021-05-25T19:10:45.774190Z","iopub.status.idle":"2021-05-25T19:10:45.915337Z","shell.execute_reply.started":"2021-05-25T19:10:45.774151Z","shell.execute_reply":"2021-05-25T19:10:45.914375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **_NOTE:_**  \nCó thể thấy rằng ta có 12 nhãn bệnh trong tập dữ liệu trên. Tuy nhiên, nhiều nhãn bệnh là sự kết hợp của các nhãn bệnh khác với nhau.  \nCho nên, thực tế ta sẽ chỉ có 5 nhãn bệnh:    \n* rust\n* scab\n* complex\n* frog_eye_leaf_spot\n* powdery_mildew\n\nVà 1 nhãn còn lại là:  \n* healthy","metadata":{}},{"cell_type":"markdown","source":"Vì **một ảnh (hay 1 lá)** có thể có **nhiều** loại **bệnh (hay nhãn bệnh)** khác nhau cho nên đây là bài toán **Multi-labels Classification!!!**","metadata":{}},{"cell_type":"code","source":"df['labels']","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.916574Z","iopub.execute_input":"2021-05-25T19:10:45.916925Z","iopub.status.idle":"2021-05-25T19:10:45.923889Z","shell.execute_reply.started":"2021-05-25T19:10:45.916878Z","shell.execute_reply":"2021-05-25T19:10:45.923110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Các nhãn đang ở dạng string cho nên chúng em tách string này ra thành list chứa các nhãn bệnh riêng biệt: ","metadata":{}},{"cell_type":"code","source":"df['labels']=df['labels'].apply( lambda string: string.split(' ') )\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.926586Z","iopub.execute_input":"2021-05-25T19:10:45.927198Z","iopub.status.idle":"2021-05-25T19:10:45.950758Z","shell.execute_reply.started":"2021-05-25T19:10:45.927154Z","shell.execute_reply":"2021-05-25T19:10:45.949772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Vì những bài toán phân loại ảnh rất quan trọng dữ liệu đầu vào vì với các ảnh chất lượng khác nhau sẽ ảnh hưởng tới kết quả của quá trình huấn luyện mô hình từ đó dẫn đến kết quả phân lớp tốt hay ko tốt.\n\nHãy cùng xem qua 1 vài ảnh cùng với nhãn và kích thước của ảnh ","metadata":{}},{"cell_type":"code","source":"train_path=\"../input/plant-pathology-2021-fgvc8/train_images\"\nplt.figure(figsize=(20,40))\ni=1\nfor idx,s in df.head(9).iterrows():\n    img_path = os.path.join(train_path,s['image'])\n    img=cv2.imread(img_path)\n    img=cv2.cvtColor(img,cv2.COLOR_BGR2RGB)\n    \n    fig=plt.subplot(9,3,i)\n    fig.imshow(img)\n    fig.set_title(s['labels'])\n    i+=1","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:45.952515Z","iopub.execute_input":"2021-05-25T19:10:45.952950Z","iopub.status.idle":"2021-05-25T19:10:56.041962Z","shell.execute_reply.started":"2021-05-25T19:10:45.952887Z","shell.execute_reply":"2021-05-25T19:10:56.041051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # 4. Đọc và sinh ảnh","metadata":{}},{"cell_type":"markdown","source":"Chúng ta có 18632 ảnh cho việc huấn luyện và chúng em chưa biết rằng liệu lượng ảnh này đã là đủ cho mô hình học mà không bị overfitting hay chưa cho nên chúng em đã thử 2 thử nghiệm:\n1. 18632 ảnh là đủ:\n- Khi thực hiện thử nghiệm này, mô hình về sau của chúng em có khả năng học rất tốt, học gần hoàn hảo 100% từ tập dữ liệu. Tuy nhiên khi chạy predict và nộp kết quả thì score lại vô cùng kém (khoảng 0.2) cho nên chúng em đã kết luận rằng thử nghiệm này là không ổn hay 18632 ảnh là không đủ để training. Và thử nghiệm thứ 2 của chúng em là sẽ Augment thêm ảnh để cho mô hình học.\n2. Augment thêm ảnh\n- Chúng em sử dụng 1 hàm có sẵn của thư viện keras đó là ImageDataGenerator để tự động augment thêm ảnh. Các tiêu chí sinh thêm ảnh được liệt kê ở block code bên dưới đây:","metadata":{}},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rescale=1/255.0, # scale giá trị của các điểm ảnh về [0, 1.0]\n    rotation_range=5, # quay ảnh 1 góc 5 rad\n    zoom_range=0.1, # phóng to, thu nhỏ ảnh trong khoảng bằng [0.1, 1.0] so với ảnh gốc\n    horizontal_flip=True, # lật ảnh theo chiều ngang\n    vertical_flip=True, # lât ảnh theo chiều dọc\n    shear_range=0.05, # làm méo ảnh ngẫu nhiên \n    brightness_range=[0.7, 1.3], # tăng giảm độ sáng của ảnh bằng [0.7, 1.3] so với ảnh gốc\n    validation_split=0.2 # chia 2 phần train và valid để training cũng như là validating model\n)\nBATCH_SIZE = 32 # batch size = 32\nHEIGHT = 224 # Chiều cao của ảnh\nWIDTH = 224 # Chiều rộng của ảnh\nCHANNEL = 3 # Số kênh màu của ảnh","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:56.043194Z","iopub.execute_input":"2021-05-25T19:10:56.043575Z","iopub.status.idle":"2021-05-25T19:10:56.049832Z","shell.execute_reply.started":"2021-05-25T19:10:56.043534Z","shell.execute_reply":"2021-05-25T19:10:56.049036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Có thể thấy các ảnh có kích thước rất lớn với độ dài, độ rộng khác nhau.  \nĐể xử lý các ảnh với kích thước lớn như này là vô cùng tốn tài nguyên cũng như là thời gian:\n- Thời gian load ảnh lên để xử lý\n- Thời gian xử lý \n- Tài nguyên CPU của kaggle là không đủ để xử lý được hết số lượng 18632 ảnh với size > 2500x2500  \nVì vậy, chúng em sử dụng 1 hàm đọc ảnh từ ImageDataGenerator để có thể load ảnh trực tiếp từ thư mục, đó là flow_from_dataframe. Thêm nữa trong hàm này chúng em lấy dữ liệu ảnh là tập ảnh đã được resize về (256,256) để quá trình chạy không tốn kém quá nhiều. ","metadata":{}},{"cell_type":"code","source":"train_generator = datagen.flow_from_dataframe(\n    df,\n    directory='../input/resized-plant2021/img_sz_256',\n    subset='training',\n    x_col='image',\n    y_col='labels',\n    target_size=(HEIGHT,WIDTH),\n    color_mode='rgb', # 3 kênh màu rgb\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=44\n    )\n\nvalid_generator = datagen.flow_from_dataframe(\n    df,\n    directory='../input/resized-plant2021/img_sz_256',\n    subset='validation',\n    x_col='image',\n    y_col='labels',\n    target_size=(HEIGHT,WIDTH),\n    color_mode='rgb',\n    class_mode='categorical',\n    batch_size=BATCH_SIZE,\n    shuffle=True,\n    seed=44\n    )","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:10:56.051031Z","iopub.execute_input":"2021-05-25T19:10:56.051565Z","iopub.status.idle":"2021-05-25T19:11:14.224103Z","shell.execute_reply.started":"2021-05-25T19:10:56.051529Z","shell.execute_reply":"2021-05-25T19:11:14.222454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # 5. Xây dựng mô hình và huấn luyện","metadata":{}},{"cell_type":"markdown","source":"Sau khi có dữ liệu ảnh, chúng em sử dụng mô hình mạng DenseNet169 làm mô hình cơ sở để huấn luyện.","metadata":{}},{"cell_type":"code","source":"weight_path='../input/keras-pretrain-model-weights/densenet169_weights_tf_dim_ordering_tf_kernels_notop.h5'","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:11:14.225473Z","iopub.execute_input":"2021-05-25T19:11:14.225821Z","iopub.status.idle":"2021-05-25T19:11:14.230250Z","shell.execute_reply.started":"2021-05-25T19:11:14.225785Z","shell.execute_reply":"2021-05-25T19:11:14.229200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Chúng em sử dụng weight của mạng mà không có các tầng fully-connected ở phía trên của mạng DenseNet","metadata":{}},{"cell_type":"markdown","source":"Chúng em xây dựng mô hình mạng neuron từ mạng DenseNet169 kết hợp với mạng fully-connected có các tầng với kích thước (256, 128)  \nTầng dense cuối cùng của mô hình sẽ có 6 unit vì đây là bài toán phân lớp với 6 nhãn. Tầng dense 6 unit này sau đấy với 1 threshold nào đó sẽ quyết định xem từng ảnh nào sẽ có 1 hay nhiều nhãn bệnh nào.","metadata":{}},{"cell_type":"code","source":"base_model=DenseNet169(weights=weight_path,include_top=False, input_shape=(HEIGHT,WIDTH,CHANNEL))\nx=base_model.output\nx=GlobalAveragePooling2D()(x)\nx=Dense(256,activation='relu')(x)\nx=Dropout(0.2)(x)\nx=Dense(128,activation='relu')(x)\npredictions=Dense(6,activation='sigmoid')(x)\n\nmodel=Model(inputs=base_model.input,outputs=predictions)\n\n# model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:13:12.644445Z","iopub.execute_input":"2021-05-25T19:13:12.644767Z","iopub.status.idle":"2021-05-25T19:13:17.665810Z","shell.execute_reply.started":"2021-05-25T19:13:12.644736Z","shell.execute_reply":"2021-05-25T19:13:17.665011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Tuy nhiên, trước khi train mô hình chúng em sẽ huấn luyện các tầng fully-connected trước.","metadata":{}},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable=False","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:13:29.711912Z","iopub.execute_input":"2021-05-25T19:13:29.712288Z","iopub.status.idle":"2021-05-25T19:13:29.735116Z","shell.execute_reply.started":"2021-05-25T19:13:29.712256Z","shell.execute_reply":"2021-05-25T19:13:29.734178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sử dụng độ đo accuracy và f1 để đánh giá","metadata":{}},{"cell_type":"code","source":"f1 = tfa.metrics.F1Score(num_classes=6,average='macro')\n\nmodel.compile(optimizer=keras.optimizers.Adam(lr=0.001), loss='binary_crossentropy',metrics=[f1])","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:13:36.172816Z","iopub.execute_input":"2021-05-25T19:13:36.173197Z","iopub.status.idle":"2021-05-25T19:13:36.203861Z","shell.execute_reply.started":"2021-05-25T19:13:36.173166Z","shell.execute_reply":"2021-05-25T19:13:36.203106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Sử dụng một vài hàm hỗ trợ việc huấn luyện như EarlyStopping và ReduceLROnPlateau để mô hình có thể học tốt hơn dựa vào việc cải thiện learning rate và khi mô hình không cải thiện được việc học thì mô hình sẽ dừng lại theo cơ chế của EarlyStopping.","metadata":{}},{"cell_type":"code","source":"earlyStopping=EarlyStopping(\n    patience=5,\n    monitor=f1,\n    mode='max',\n    restore_best_weights=True\n)\nlrSchedule = ReduceLROnPlateau(\n    monitor='val_f1_score', \n    factor=0.05, \n    patience=4, \n    verbose=1\n)","metadata":{"execution":{"iopub.status.busy":"2021-05-25T19:14:41.384153Z","iopub.execute_input":"2021-05-25T19:14:41.384479Z","iopub.status.idle":"2021-05-25T19:14:41.388576Z","shell.execute_reply.started":"2021-05-25T19:14:41.384448Z","shell.execute_reply":"2021-05-25T19:14:41.387757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Huấn luyện mô hình","metadata":{}},{"cell_type":"code","source":"hist = model.fit_generator(\n    generator=train_generator,\n    validation_data=valid_generator,\n    epochs=30,\n    steps_per_epoch=train_generator.samples//128,\n    validation_steps=valid_generator.samples//128,\n    callbacks=[earlyStopping, lrSchedule]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for i, layer in enumerate(model.layers):\n#     print(i, layer.name, \"-\", layer.trainable)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.layers[595:]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"freeze các tầng đã train từ trước lại và tiếp tục train các tầng còn lại","metadata":{}},{"cell_type":"code","source":"for layer in model.layers:\n    layer.trainable=True\n\nmodel.compile(optimizer=keras.optimizers.Adam(lr=0.001), loss='binary_crossentropy',metrics=[f1])\nhistory = model.fit_generator(generator=train_generator,\n                    validation_data=valid_generator,\n                    epochs=30,\n                    steps_per_epoch=train_generator.samples//128,\n                    validation_steps=valid_generator.samples//128,\n                    callbacks=[earlyStopping, lrSchedule])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Biểu thị lại điểm accuracy theo từng epoch","metadata":{}},{"cell_type":"code","source":"# plt.figure(figsize=(15,6))\n# epoch_list = list(range(1, len(history.history['accuracy']) + 1))\n# plt.plot(epoch_list, history.history['accuracy'],label='accuracy')\n# plt.plot(epoch_list, history.history['val_accuracy'],label='val_accuracy')\n# plt.xlabel('epoches')\n# plt.ylabel('accuracy')\n# plt.legend()\n# plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Biểu thị lại điểm f1 theo từng epoch","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,6))\nepoch__list = list(range(1,len(history.history['f1_score'])+1))\nplt.plot(epoch__list, history.history['f1_score'],label='f1_score')\nplt.plot(epoch__list, history.history['val_f1_score'],label='val_f1_score')\nplt.xlabel('epoches')\nplt.ylabel('f1')\nplt.legend()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('plant_densenet169_ver02.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # 6. Dự đoán trên tập test_images và nộp bài","metadata":{}},{"cell_type":"markdown","source":"Đọc dữ liệu file sample_submission","metadata":{}},{"cell_type":"code","source":"sample_sub = pd.read_csv('../input/plant-pathology-2021-fgvc8/sample_submission.csv')\nsample_sub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lấy ảnh từ file test_images để dự đoán","metadata":{}},{"cell_type":"code","source":"test_data = datagen.flow_from_dataframe(\n    sample_sub,\n    directory='../input/plant-pathology-2021-fgvc8/test_images',\n    x_col='image',\n    y_col=None,\n    color_mode='rgb',\n    target_size=(HEIGHT,WIDTH),\n    class_mode=None,\n    shuffle=False\n)\n\npredictions = model.predict(test_data)\nprint(predictions)\n\nclass_idx=[]\nfor pred in predictions:\n    pred=list(pred)\n    temp=[]\n    for i in pred:\n        if (i>0.3):\n            temp.append(pred.index(i))\n    if (temp!=[]):\n        class_idx.append(temp)\n    else:\n        temp.append(np.argmax(pred))\n        class_idx.append(temp)\nprint(class_idx)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In ra kết quả dự đoán","metadata":{}},{"cell_type":"code","source":"class_dict = train_generator.class_indices\ndef get_key(val):\n    for key,value in class_dict.items():\n        if (val==value):\n            return key\nprint(class_dict)\n\nsub_pred=[]\nfor img_ in class_idx:\n    img_pred=[]\n    for i in img_:\n        img_pred.append(get_key(i))\n    sub_pred.append( ' '.join(img_pred))\nprint(sub_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sample_sub[['image']]\nsub['labels']=sub_pred\nsub","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In ra file submission.csv để nộp bài","metadata":{}},{"cell_type":"code","source":"sub.to_csv('submission.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}