{"cells":[{"metadata":{},"cell_type":"markdown","source":"Ref: https://www.youtube.com/watch?v=8J5Q4mEzRtY&list=PL98nY_tJQXZntH5WUtKB0bghZeKVIJHJc","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"* First build a good cross validation system","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Create folds\nMultilabel classification problem, we can have multiple model, or a single model can do the job\nFor multilabel classification, we are using Multilabel Stratified KFold","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install iterative-stratification","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom iterstrat.ml_stratifiers import MultilabelStratifiedKFold\n\nimport numpy as np\nimport joblib\nimport glob\nfrom tqdm import tqdm\n\nfrom zipfile import ZipFile\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"if __name__ == \"__main__\":\n    df = pd.read_csv(\"../input/bengaliai-cv19/train.csv\")\n    print(df.head())\n    df.loc[:, 'kfold'] = -1\n    \n    #shuffling dataset\n    #frac = ?\n    df = df.sample(frac = 1).reset_index(drop = True)\n    \n    x = df.image_id.values\n    y = df[[\"grapheme_root\", \"vowel_diacritic\", \"consonant_diacritic\"]].values\n    \n    mskf = MultilabelStratifiedKFold(n_splits = 5)\n    \n    for fold, (trn_, val_) in enumerate(mskf.split(x, y)):\n        print(\"TRAIN: \", trn_, \"VAL: \",  val_)\n        df.loc[val_, \"kfold\"] = fold\n        \n    print(df.kfold.value_counts())\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1 = pd.read_parquet(\"../input/bengaliai-cv19/train_image_data_0.parquet\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_1.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Pickles","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"!mkdir image_pickles.zip","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nif __name__ == \"__main__\":\n    files = glob.glob(\"../input/bengaliai-cv19/train_*.parquet\")\n    folder = \"..output/kaggle/working/image_pickles.zip\"\n    zipObj = ZipFile(folder, \"w\")\n    for f in files:\n        df = pd.read_parquet(f)\n        image_ids = df.image_id.values\n        df = df.drop(\"image_id\", axis = 1)\n        image_array = df.values\n        for j, image_id in tqdm(enumerate(image_ids), total = len(image_ids)):\n            label = \"image_pickles\" + image_id + \".pkl\"\n            joblib.dump(image_array[j, :], label)\n            zipObj.write(label)\n    zipObj.close()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Yeah Kaggle :/ whatever","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Model\n* three different model\n* one model\n\nFirst create a dataset class","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"class BengaliDatasetTrain:\n    def __init__(self, folds, img_height, img_width, mean, std):\n        df = df\n        df = df[[\"image_id\", \"grapheme_root\", \"vowel_diacritic\", \"consonent_diacritic\", \"kfold\"]]\n        \n        df = df[df.kfold.isin(folds)].reset_index(drop = True)\n        self.image_ids = df.image_id.values\n        self.grapheme_root = df.grapheme_root.values\n        self.vowel_diacritic = df.vowel_diacritic.values\n        self.consonent_diacritic = df.consonent_diacritic.values\n        \n    def __len__(self):\n        return len(self.image_ids)\n    \n    def __getitem__(self, item):\n        image = joblib.load(\"..input/image_pickles/{self.image_ids[item]}.pkl\")\n        #image is vector\n        image = image.reshape(137, 236).astype(float)\n        \n    \n    ","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}