{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# بخش اول) آماده سازی محیط\nدر این بخش محتوای فولدر حاوی فایل‌های ورودی این پروژه نیز نمایش داده شده است.","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\n\nimport sys\nsys.path.append('../input/bert-baseline-pre-and-post-process/')\n\nimport preprocessv5 as preprocess  #contains preprocessing util functions\nimport postprocessv6 as postprocess #contains postprocessing util functions\nimport to_pklv5 as to_pkl\nimport pkl_to_tfrecordsv5 as pkl_to_tfrecords\n\nimport json  # for json file usage\nimport tqdm  # for progress bar visualization\n\nimport absl\n\nimport os\n# show input folder directories and files\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:09.379311Z","iopub.execute_input":"2021-06-18T17:32:09.379663Z","iopub.status.idle":"2021-06-18T17:32:15.103466Z","shell.execute_reply.started":"2021-06-18T17:32:09.379610Z","shell.execute_reply":"2021-06-18T17:32:15.102831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.getpid()","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:15.106105Z","iopub.execute_input":"2021-06-18T17:32:15.106402Z","iopub.status.idle":"2021-06-18T17:32:15.116402Z","shell.execute_reply.started":"2021-06-18T17:32:15.106355Z","shell.execute_reply":"2021-06-18T17:32:15.115604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# بخش دوم) تنظیمات اولیه","metadata":{}},{"cell_type":"code","source":"use_wta = True # set the activasion function to select best answers to WTA strategy\n\n# get input files and paths\non_kaggle_server = os.path.exists('/kaggle')\nnq_test_file = '../input/tensorflow2-question-answering/simplified-nq-test.jsonl' \npublic_dataset = os.path.getsize(nq_test_file)<20_000_000\nprivate_dataset = os.path.getsize(nq_test_file)>=20_000_000\nmodel_path = '../input/tpu-2020-01-22/'\n\n# show above variables\nfor k in ['on_kaggle_server','nq_test_file','public_dataset','private_dataset']:\n    print(k,globals()[k],sep=': ')","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:15.117837Z","iopub.execute_input":"2021-06-18T17:32:15.118152Z","iopub.status.idle":"2021-06-18T17:32:15.127822Z","shell.execute_reply.started":"2021-06-18T17:32:15.118105Z","shell.execute_reply":"2021-06-18T17:32:15.126760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"مدل «برت» از پیش آموزش دیده شده بارگزاری می‌شود","metadata":{}},{"cell_type":"code","source":"# load pretrained model as a tensorflow model\nmodel = tf.saved_model.load(model_path)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:15.130151Z","iopub.execute_input":"2021-06-18T17:32:15.130556Z","iopub.status.idle":"2021-06-18T17:32:40.624422Z","shell.execute_reply.started":"2021-06-18T17:32:15.130480Z","shell.execute_reply":"2021-06-18T17:32:40.623535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# create a vocab file in the \".../model/assets/vocab-nq.txt\" which contains all words in all files\nto_pkl.jsonl_to_pkl(source=nq_test_file,output='features.pkl',\n                vocab=model_path +'assets/vocab-nq.txt',\n                max_contexts=-1,lower_case=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:40.627403Z","iopub.execute_input":"2021-06-18T17:32:40.627683Z","iopub.status.idle":"2021-06-18T17:32:54.353877Z","shell.execute_reply.started":"2021-06-18T17:32:40.627639Z","shell.execute_reply":"2021-06-18T17:32:54.353121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"در این بخش تمامی کلمات موجود در داده‌های دیتاست به صورت یک دیکشنری در یک متغیر و یک فایل ذخیره می‌گردد.","metadata":{}},{"cell_type":"code","source":"# create tensor flow records based on the vocab file (with coresponding metadata metrics)\npkl_to_tfrecords._convert(source='features.pkl',output='all.tfrecords',meta_data='meta_data',shuffle=False,shuffle_size=0,yield_segment_variant='nolabels')","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:54.355809Z","iopub.execute_input":"2021-06-18T17:32:54.356089Z","iopub.status.idle":"2021-06-18T17:32:57.236690Z","shell.execute_reply.started":"2021-06-18T17:32:54.356045Z","shell.execute_reply":"2021-06-18T17:32:57.235660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# بخش سوم) پردازش داده\nپیش پردازش داده در این تابع انجام می‌شود","metadata":{}},{"cell_type":"code","source":"def input_fn(input_file_pattern,seq_length=512,batch_size=4):\n    def mk_labels(ex):          \n        qlen = ex.pop('question_len')\n        dlen = ex.pop('data_len')\n        input_mask = tf.sequence_mask(dlen,seq_length,dtype=tf.int32)\n        ex['input_mask']  = input_mask\n        ex['segment_ids'] = tf.minimum(input_mask,1-tf.sequence_mask(qlen,seq_length,dtype=tf.int32))\n        return ex\n\n    name_to_features = {\n        'input_ids'   : tf.io.FixedLenFeature([seq_length], tf.int64),\n        'question_len': tf.io.FixedLenFeature([], tf.int64),\n        'data_len'    : tf.io.FixedLenFeature([], tf.int64),\n    }\n    name_to_features['unique_id']   = tf.io.FixedLenFeature([2], tf.int64)\n\n    def decode(record):\n        ex = tf.io.parse_single_example(record, name_to_features)\n        for k,v in ex.items():\n            if k!='unique_id':\n                ex[k] = tf.cast(v,tf.int32)\n        return ex\n\n    input_files = tf.io.gfile.glob(input_file_pattern)        \n    d = tf.data.TFRecordDataset(input_files)\n    d = d.map(decode)\n    d = d.batch(batch_size,drop_remainder=False)\n    #d = d.map(mk_labels)\n    d = d.prefetch(128)\n    return d\n","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:57.238139Z","iopub.execute_input":"2021-06-18T17:32:57.238635Z","iopub.status.idle":"2021-06-18T17:32:57.251574Z","shell.execute_reply.started":"2021-06-18T17:32:57.238436Z","shell.execute_reply":"2021-06-18T17:32:57.250617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"وظیفه این تابع پس پردازش اولیه خروجی مدل است","metadata":{}},{"cell_type":"code","source":"def output_fn():\n    def _output_fn(unique_id,model_output,n_keep=100):\n        pos_logits,ans_logits,long_mask,short_mask,cross = model_output\n\n        long_span_logits =  pos_logits\n        mask = tf.cast(tf.expand_dims(long_mask,-1),long_span_logits.dtype)\n\n        long_span_logits = long_span_logits-10000*mask \n        long_p = tf.nn.softmax(long_span_logits,axis=1)\n\n        short_span_logits = pos_logits\n        short_span_logits -= 10000*tf.cast(tf.expand_dims(short_mask,-1),short_span_logits.dtype)\n        start_logits,end_logits = short_span_logits[:,:,0],short_span_logits[:,:,1]\n\n        batch_size,seq_length = short_span_logits.shape[0],short_span_logits.shape[1]\n        seq = tf.range(seq_length)\n        i_leq_j_mask = tf.cast(tf.expand_dims(seq,1)>tf.expand_dims(seq,0),short_span_logits.dtype)\n        i_leq_j_mask = tf.expand_dims(i_leq_j_mask,0)\n\n        logits  = tf.expand_dims(start_logits,2)+tf.expand_dims(end_logits,1)+cross\n        logits -= 10000*i_leq_j_mask\n        logits  = tf.reshape(logits, [batch_size,seq_length*seq_length])\n        short_p = tf.nn.softmax(logits)\n        indices = tf.argsort(short_p,axis=1,direction='DESCENDING')[:,:n_keep]\n        short_p = tf.gather(short_p,indices,batch_dims=1)\n\n        return dict(unique_id = unique_id,\n                    ans_logits= ans_logits,\n                    long_p    = long_p,\n                    short_p   = short_p,\n                    short_p_indices = indices)\n    return _output_fn\n","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:57.253017Z","iopub.execute_input":"2021-06-18T17:32:57.253527Z","iopub.status.idle":"2021-06-18T17:32:57.268947Z","shell.execute_reply.started":"2021-06-18T17:32:57.253339Z","shell.execute_reply":"2021-06-18T17:32:57.268141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# بخش چهارم) اعمال مدل بر روی دیتاست\nدر این بخش اطلاعت ورودی به صورت یک فایل رکورد های تنسور فلو خوانده می‌شود و خروجی مدل (حاصل از پردازش فایل ورودی) آماده پس پردازش می‌شود و خروجی نهایی بدست می‌آید.","metadata":{}},{"cell_type":"code","source":"d = input_fn('all.tfrecords',batch_size=64) \nif public_dataset:\n    d = d.take(3)\nif not on_kaggle_server:\n    d = tqdm.notebook.tqdm(d)\nresults = []\noutput = output_fn() \nfor b in d:\n    unique_id = b.pop('unique_id').numpy()\n    b = [b['data_len'],b['input_ids'],b['question_len']]\n    # print(b.keys())\n    #pos_logits,ans_logits,mask_0,mask_1 = \n    out_dict = output(unique_id,model(b,training=False))\n    for k,v in out_dict.items():\n            if isinstance(v,tf.Tensor):\n                out_dict[k] = v.numpy()\n    results.append(out_dict)\n\nraw_results = postprocess.read_rawresult(results)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:32:57.272165Z","iopub.execute_input":"2021-06-18T17:32:57.272407Z","iopub.status.idle":"2021-06-18T17:33:12.561736Z","shell.execute_reply.started":"2021-06-18T17:32:57.272364Z","shell.execute_reply":"2021-06-18T17:33:12.560883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"d","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.563052Z","iopub.execute_input":"2021-06-18T17:33:12.563346Z","iopub.status.idle":"2021-06-18T17:33:12.574378Z","shell.execute_reply.started":"2021-06-18T17:33:12.563302Z","shell.execute_reply":"2021-06-18T17:33:12.573686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# an example of raw results shown here\n# every dataset record is splited to multiple records with window size of 512 and stride of 128\nprint(results)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-06-18T17:33:12.575950Z","iopub.execute_input":"2021-06-18T17:33:12.576752Z","iopub.status.idle":"2021-06-18T17:33:12.604442Z","shell.execute_reply.started":"2021-06-18T17:33:12.576565Z","shell.execute_reply":"2021-06-18T17:33:12.603839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"اطلاعات مفید موجود در خروجی مدل بررسی می‌شوند","metadata":{}},{"cell_type":"code","source":"iterator = postprocess.pickle_iter('features.pkl')\nif not on_kaggle_server:\n    iterator = tqdm.notebook.tqdm(iterator)\nrecords = postprocess.read_features(iterator)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.605811Z","iopub.execute_input":"2021-06-18T17:33:12.606220Z","iopub.status.idle":"2021-06-18T17:33:12.696007Z","shell.execute_reply.started":"2021-06-18T17:33:12.606047Z","shell.execute_reply":"2021-06-18T17:33:12.695294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"در دو بخش زیر با استفاده از استراتژی «برنده به جا» جواب متناظر با هر رکورد مشخص می‌شود.","metadata":{}},{"cell_type":"code","source":"examples = postprocess.compute_examples(raw_results,records)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.697319Z","iopub.execute_input":"2021-06-18T17:33:12.697785Z","iopub.status.idle":"2021-06-18T17:33:12.703948Z","shell.execute_reply.started":"2021-06-18T17:33:12.697682Z","shell.execute_reply":"2021-06-18T17:33:12.702795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#e2p = postprocess.ExampleToProb(keep_threshold=0.1,null_prob_threshold=1e-4)\nSummary = postprocess.WTASummary #if use_wta else postprocessv3.ProbSummary\nsummary = Summary(min_vote_prob=0.1)\npredictions = [summary(e) for e in tqdm.notebook.tqdm(examples)]","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.705312Z","iopub.execute_input":"2021-06-18T17:33:12.705845Z","iopub.status.idle":"2021-06-18T17:33:12.793901Z","shell.execute_reply.started":"2021-06-18T17:33:12.705672Z","shell.execute_reply":"2021-06-18T17:33:12.793177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"نتیجه اجرای برنامه و جواب متناظر با هر رکورد در فایل سی اس وی ذخیره می‌شود.","metadata":{}},{"cell_type":"code","source":"index = pd.read_csv('../input/tensorflow2-question-answering/sample_submission.csv').example_id\n\nsubmission = postprocess.create_submission_df(predictions,index=index,\n                                                long_threshold=0.94 if use_wta else 0.77,\n                                                short_threshold=0.94 if use_wta else 0.77 ,\n                                                yes_no_threshold=0.6)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.795194Z","iopub.execute_input":"2021-06-18T17:33:12.795690Z","iopub.status.idle":"2021-06-18T17:33:12.827782Z","shell.execute_reply.started":"2021-06-18T17:33:12.795643Z","shell.execute_reply":"2021-06-18T17:33:12.827039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.829053Z","iopub.execute_input":"2021-06-18T17:33:12.829567Z","iopub.status.idle":"2021-06-18T17:33:12.979459Z","shell.execute_reply.started":"2021-06-18T17:33:12.829519Z","shell.execute_reply":"2021-06-18T17:33:12.978724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:12.980851Z","iopub.execute_input":"2021-06-18T17:33:12.981132Z","iopub.status.idle":"2021-06-18T17:33:12.998538Z","shell.execute_reply.started":"2021-06-18T17:33:12.981087Z","shell.execute_reply":"2021-06-18T17:33:12.997733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(50) # longer version","metadata":{"execution":{"iopub.status.busy":"2021-06-18T17:33:13.000143Z","iopub.execute_input":"2021-06-18T17:33:13.000678Z","iopub.status.idle":"2021-06-18T17:33:13.014586Z","shell.execute_reply.started":"2021-06-18T17:33:13.000414Z","shell.execute_reply":"2021-06-18T17:33:13.013717Z"},"trusted":true},"execution_count":null,"outputs":[]}]}