{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-22T21:43:26.271543Z","iopub.execute_input":"2021-12-22T21:43:26.271920Z","iopub.status.idle":"2021-12-22T21:44:43.664387Z","shell.execute_reply.started":"2021-12-22T21:43:26.271870Z","shell.execute_reply":"2021-12-22T21:44:43.663473Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install elasticsearch[async] pandasticsearch\n!pip install hashedindex\n!apt-get update\n!apt-get install tesseract-ocr\n!apt-get install tesseract-ocr-por\n!pip install tika","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:44:43.666745Z","iopub.execute_input":"2021-12-22T21:44:43.667070Z","iopub.status.idle":"2021-12-22T21:45:17.953374Z","shell.execute_reply.started":"2021-12-22T21:44:43.667013Z","shell.execute_reply":"2021-12-22T21:45:17.952292Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import hashedindex\nimport tika\nimport requests\nfrom tika import parser\nfrom hashedindex import textparser\n\n\nindex = hashedindex.HashedIndex()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:17.955122Z","iopub.execute_input":"2021-12-22T21:45:17.955366Z","iopub.status.idle":"2021-12-22T21:45:17.961387Z","shell.execute_reply.started":"2021-12-22T21:45:17.955336Z","shell.execute_reply":"2021-12-22T21:45:17.960343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topics_rnd3 = pd.read_csv('/kaggle/input/trec-covid-information-retrieval/topics-rnd3.csv')\nmetadata = pd.read_csv('/kaggle/input/trec-covid-information-retrieval/CORD-19/CORD-19/metadata.csv')","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:17.962916Z","iopub.execute_input":"2021-12-22T21:45:17.963215Z","iopub.status.idle":"2021-12-22T21:45:21.024335Z","shell.execute_reply.started":"2021-12-22T21:45:17.963175Z","shell.execute_reply":"2021-12-22T21:45:21.023314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.027939Z","iopub.execute_input":"2021-12-22T21:45:21.028433Z","iopub.status.idle":"2021-12-22T21:45:21.055930Z","shell.execute_reply.started":"2021-12-22T21:45:21.028378Z","shell.execute_reply":"2021-12-22T21:45:21.054955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata.info()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.057839Z","iopub.execute_input":"2021-12-22T21:45:21.058151Z","iopub.status.idle":"2021-12-22T21:45:21.271654Z","shell.execute_reply.started":"2021-12-22T21:45:21.058114Z","shell.execute_reply":"2021-12-22T21:45:21.270600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topics_rnd3.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.273331Z","iopub.execute_input":"2021-12-22T21:45:21.273639Z","iopub.status.idle":"2021-12-22T21:45:21.286166Z","shell.execute_reply.started":"2021-12-22T21:45:21.273603Z","shell.execute_reply":"2021-12-22T21:45:21.285228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"topics_rnd3.info()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.288826Z","iopub.execute_input":"2021-12-22T21:45:21.289702Z","iopub.status.idle":"2021-12-22T21:45:21.305012Z","shell.execute_reply.started":"2021-12-22T21:45:21.289651Z","shell.execute_reply":"2021-12-22T21:45:21.303960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset = pd.DataFrame(metadata, columns = ['cord_uid', 'title', 'abstract', 'pdf_json_files'])\nsubset_topics = pd.DataFrame(topics_rnd3, columns = ['topic-id', 'query'])","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.306785Z","iopub.execute_input":"2021-12-22T21:45:21.307019Z","iopub.status.idle":"2021-12-22T21:45:21.323752Z","shell.execute_reply.started":"2021-12-22T21:45:21.306991Z","shell.execute_reply":"2021-12-22T21:45:21.322754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.325411Z","iopub.execute_input":"2021-12-22T21:45:21.326243Z","iopub.status.idle":"2021-12-22T21:45:21.345866Z","shell.execute_reply.started":"2021-12-22T21:45:21.326202Z","shell.execute_reply":"2021-12-22T21:45:21.344602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_topics.head()","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.349591Z","iopub.execute_input":"2021-12-22T21:45:21.350269Z","iopub.status.idle":"2021-12-22T21:45:21.362657Z","shell.execute_reply.started":"2021-12-22T21:45:21.350225Z","shell.execute_reply":"2021-12-22T21:45:21.361509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Start running the tika service\ntika.initVM()\n\nfinal = []\nsize = 20000","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.364393Z","iopub.execute_input":"2021-12-22T21:45:21.364790Z","iopub.status.idle":"2021-12-22T21:45:21.374480Z","shell.execute_reply.started":"2021-12-22T21:45:21.364758Z","shell.execute_reply":"2021-12-22T21:45:21.373799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headers = {\n    \"X-Tika-OCRLanguage\": \"eng\"\n}\ndef catch_file(data):\n    return parser.from_file(\"/kaggle/input/trec-covid-information-retrieval/CORD-19/CORD-19/\" + data[\"pdf_json_files\"], headers=headers)","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.376574Z","iopub.execute_input":"2021-12-22T21:45:21.377207Z","iopub.status.idle":"2021-12-22T21:45:21.385468Z","shell.execute_reply.started":"2021-12-22T21:45:21.377160Z","shell.execute_reply":"2021-12-22T21:45:21.384432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for index_t, topic in subset_topics.iterrows():\n    for index_m, data in subset.head(size).iterrows():\n        if isinstance(data.pdf_json_files, str):\n            try:\n                text = \"\"\n                conteudo = catch_file(data)\n                #print(conteudo.keys())\n                for x in conteudo[\"content\"].splitlines():\n                    if \"\\\"text\\\"\" in line or \"\\\"title\\\"\" in x:\n                        text += x\n                \n                text = text.replace(\"\\\"text\\\":\", \"\").replace(\"\\\"title\\\":\", \"\").replace(\"\\\"\", \"\")\n                \n                count = text.count(topic.query)\n                \n                final.append([topic[\"topic-id\"], data[\"cord_uid\"], count]) \n            except:\n                continue\n#final","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.389670Z","iopub.execute_input":"2021-12-22T21:45:21.389937Z","iopub.status.idle":"2021-12-22T21:45:21.766405Z","shell.execute_reply.started":"2021-12-22T21:45:21.389896Z","shell.execute_reply":"2021-12-22T21:45:21.765553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = pd.DataFrame(final, columns = ['topic-id', 'cord-id', 'count'])\nresult = result.sort_values(by='topic-id', ascending=True)\nresult.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.767745Z","iopub.execute_input":"2021-12-22T21:45:21.767984Z","iopub.status.idle":"2021-12-22T21:45:21.782372Z","shell.execute_reply.started":"2021-12-22T21:45:21.767947Z","shell.execute_reply":"2021-12-22T21:45:21.781408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result[[\"topic-id\",\"cord-id\"]].to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2021-12-22T21:45:21.783735Z","iopub.execute_input":"2021-12-22T21:45:21.784001Z","iopub.status.idle":"2021-12-22T21:45:21.808821Z","shell.execute_reply.started":"2021-12-22T21:45:21.783968Z","shell.execute_reply":"2021-12-22T21:45:21.807771Z"},"trusted":true},"execution_count":null,"outputs":[]}]}