{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"! pip install 'git+https://github.com/PyTorchLightning/lightning-flash.git#egg=lightning-flash[text]' -q","metadata":{"execution":{"iopub.status.busy":"2021-12-17T15:03:19.974238Z","iopub.execute_input":"2021-12-17T15:03:19.974586Z","iopub.status.idle":"2021-12-17T15:04:05.818320Z","shell.execute_reply.started":"2021-12-17T15:03:19.974510Z","shell.execute_reply":"2021-12-17T15:04:05.817472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from flash import Trainer\nfrom flash.text import QuestionAnsweringData, QuestionAnsweringTask\nfrom flash.text.question_answering.input import QuestionAnsweringInputBase, QuestionAnsweringDictionaryInput\nfrom datasets import Dataset, load_dataset\nimport pandas as pd\nimport json\nfrom typing import Union","metadata":{"execution":{"iopub.status.busy":"2021-12-17T15:04:05.820228Z","iopub.execute_input":"2021-12-17T15:04:05.820499Z","iopub.status.idle":"2021-12-17T15:04:16.931968Z","shell.execute_reply.started":"2021-12-17T15:04:05.820463Z","shell.execute_reply":"2021-12-17T15:04:16.931172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(entry, short_answer=True):\n    question = entry['question_text']\n    text = entry['document_text'].split(' ')\n    annotations = entry['annotations'][0]\n    id_ = entry['example_id']\n\n    for i, candidate in enumerate(entry['long_answer_candidates']):\n        isThereIndex = True if i == annotations['long_answer']['candidate_index'] else False\n        long_start = candidate['start_token']\n        long_end = candidate['end_token']\n        if isThereIndex:\n            short_start = 0 \n            short_end = 0\n            if len(annotations['short_answers']) > 0:\n                short_start = annotations['short_answers'][0]['start_token']\n                short_end = annotations['short_answers'][0]['end_token']\n\n                short_start = short_start - long_start\n                short_end = short_end - long_start\n            long_answer = ' '.join(text[long_start:long_end])\n            short_answer = ' '.join(long_answer.split(' ')[short_start:short_end])\n            if short_answer:\n                return (id_, '', ' '.join(text), question, {\"text\": short_answer, \"answer_start\": [short_start]})\n            else:\n                return (id_, '', ' '.join(text), question, {\"text\": long_answer, \"answer_start\": [long_start]})","metadata":{"execution":{"iopub.status.busy":"2021-12-17T15:04:16.933261Z","iopub.execute_input":"2021-12-17T15:04:16.933504Z","iopub.status.idle":"2021-12-17T15:04:16.950807Z","shell.execute_reply.started":"2021-12-17T15:04:16.933473Z","shell.execute_reply":"2021-12-17T15:04:16.949526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class QuestionAnsweringTFInput(QuestionAnsweringInputBase):\n    \n    num_samples: int = 1000\n\n    def load_data(\n        self,\n        json_file,\n        field: str,\n        max_source_length: int = 384,\n        max_target_length: int = 30,\n        padding: Union[str, bool] = \"max_length\",\n        question_column_name: str = \"question\",\n        context_column_name: str = \"context\",\n        answer_column_name: str = \"answer\",\n        doc_stride: int = 128,\n    ):\n#        dataset_dict = load_dataset(\"json\", data_files={\"data\": str(json_file)})\n        ids = []\n        titles = []\n        contexts = []\n        questions = []\n        answers = []\n        with open(json_file) as f:\n            for i in range(self.num_samples):\n                line = json.loads(f.readline())\n                # line = json.loads(line)\n                result = process_data(line, short_answer=True)\n                if result:\n                    id_, title, context, question, answer = result\n                    ids.append(id_)\n                    titles.append(title)\n                    contexts.append(context)\n                    questions.append(question)\n                    answers.append(answer)\n\n        data = {\"id\": ids, \"title\": titles, \"context\": contexts, \"question\": questions, \"answer\": answers}\n        \n        return super().load_data(\n            Dataset.from_dict(data),\n            max_source_length=max_source_length,\n            max_target_length=max_target_length,\n            padding=padding,\n            question_column_name=question_column_name,\n            context_column_name=context_column_name,\n            answer_column_name=answer_column_name,\n            doc_stride=doc_stride,\n        )","metadata":{"execution":{"iopub.status.busy":"2021-12-17T15:04:16.952608Z","iopub.execute_input":"2021-12-17T15:04:16.952848Z","iopub.status.idle":"2021-12-17T15:04:16.969347Z","shell.execute_reply.started":"2021-12-17T15:04:16.952812Z","shell.execute_reply":"2021-12-17T15:04:16.968496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datamodule = QuestionAnsweringData.from_json(\n    train_file='../input/tensorflow2-question-answering/simplified-nq-train.jsonl',\n    input_cls=QuestionAnsweringTFInput,\n    batch_size=1,\n    val_split=0.1,\n)\n\n\n# 2. Build the task\nmodel = QuestionAnsweringTask(backbone='prajjwal1/bert-tiny')\n\n# 3. Create the trainer and finetune the model\ntrainer = Trainer(max_epochs=1, gpus=1, precision=16, limit_train_batches=10, limit_val_batches=0)\ntrainer.fit(model, datamodule=datamodule)\ntrainer.validate(model, datamodule)","metadata":{"execution":{"iopub.status.busy":"2021-12-17T15:04:56.356158Z","iopub.execute_input":"2021-12-17T15:04:56.356762Z","iopub.status.idle":"2021-12-17T15:05:01.081422Z","shell.execute_reply.started":"2021-12-17T15:04:56.356722Z","shell.execute_reply":"2021-12-17T15:05:01.080687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}