{"cells":[{"metadata":{},"cell_type":"markdown","source":"# Rewrite [nofreewill](https://www.kaggle.com/nofreewill)'s [notebook](https://www.kaggle.com/nofreewill/normalize-your-predictions) and make it can run on Kaggle.\n\n!!! The LB score may get worse."},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2021-03-31T03:14:29.239029Z","iopub.status.busy":"2021-03-31T03:14:29.238359Z","iopub.status.idle":"2021-03-31T03:15:51.797199Z","shell.execute_reply":"2021-03-31T03:15:51.796472Z"},"papermill":{"duration":82.568173,"end_time":"2021-03-31T03:15:51.797457","exception":false,"start_time":"2021-03-31T03:14:29.229284","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"!conda install -y -c rdkit rdkit","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-31T03:15:52.007865Z","iopub.status.busy":"2021-03-31T03:15:52.006931Z","iopub.status.idle":"2021-03-31T03:15:52.011057Z","shell.execute_reply":"2021-03-31T03:15:52.010316Z"},"papermill":{"duration":0.112062,"end_time":"2021-03-31T03:15:52.011221","exception":false,"start_time":"2021-03-31T03:15:51.899159","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"%%writefile normalize_inchis.py\n\nfrom tqdm import tqdm\nfrom rdkit import Chem\nfrom rdkit import RDLogger\nRDLogger.DisableLog('rdApp.*')\nfrom pathlib import Path\n\ndef normalize_inchi(inchi):\n    try:\n        mol = Chem.MolFromInchi(inchi)\n    except:\n        pass\n    if mol is None:\n        return inchi\n    else:\n        try: return Chem.MolToInchi(mol)\n        except: return inchi\n        \nsubmission_name = '../input/bmsmt-ds-model/submission_3.06.csv'\nnorm_path = Path('submission_norm.csv')\n\n# Do the job\nN = norm_path.read_text().count('\\n') if norm_path.exists() else 0\nprint(N, 'number of predictions already normalized')\n\nr = open(submission_name, 'r')\nwrite_mode = 'w' if N == 0 else 'a'\nw = open(str(norm_path), write_mode, buffering=1)\n\nfor _ in range(N):\n    r.readline()\nline = r.readline()  # this line is the header or is where it died last time\nw.write(line)\n\npbar = tqdm()\nwhile True:\n    line = r.readline()\n    if not line:\n        break  # done\n    image_id = line.split(',')[0]\n    inchi = ','.join(line[:-1].split(',')[1:]).replace('\"','')\n    inchi_norm = normalize_inchi(inchi)\n    w.write(f'{image_id},\"{inchi_norm}\"\\n')\n    pbar.update(1)\n\nr.close()\nw.close()","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-31T03:15:52.217514Z","iopub.status.busy":"2021-03-31T03:15:52.213924Z","iopub.status.idle":"2021-03-31T03:15:52.942801Z","shell.execute_reply":"2021-03-31T03:15:52.942104Z"},"papermill":{"duration":0.832068,"end_time":"2021-03-31T03:15:52.942948","exception":false,"start_time":"2021-03-31T03:15:52.11088","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"!ls","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-31T03:15:53.165849Z","iopub.status.busy":"2021-03-31T03:15:53.164653Z","iopub.status.idle":"2021-03-31T03:36:11.417061Z","shell.execute_reply":"2021-03-31T03:36:11.417655Z"},"papermill":{"duration":1218.370652,"end_time":"2021-03-31T03:36:11.417885","exception":false,"start_time":"2021-03-31T03:15:53.047233","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"!while [ 1 ]; do python normalize_inchis.py && break; done","execution_count":null,"outputs":[]},{"metadata":{"execution":{"iopub.execute_input":"2021-03-31T03:36:18.442482Z","iopub.status.busy":"2021-03-31T03:36:18.441836Z","iopub.status.idle":"2021-03-31T03:36:18.443985Z","shell.execute_reply":"2021-03-31T03:36:18.444638Z"},"papermill":{"duration":3.50649,"end_time":"2021-03-31T03:36:18.444841","exception":false,"start_time":"2021-03-31T03:36:14.938351","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"# import pandas as pd\n# import Levenshtein\n# from tqdm import tqdm\n\n# sub_df = pd.read_csv(submission_name)\n# sub_norm_df = pd.read_csv(norm_path)\n\n# lev = 0\n# N = len(sub_df)\n# for i in tqdm(range(N)):\n#     inchi, inchi_norm = sub_df.iloc[i,1], sub_norm_df.iloc[i,1]\n#     lev += Levenshtein.distance(inchi, inchi_norm)\n\n# print(lev/N)","execution_count":null,"outputs":[]},{"metadata":{"papermill":{"duration":3.43815,"end_time":"2021-03-31T03:36:25.3238","exception":false,"start_time":"2021-03-31T03:36:21.88565","status":"completed"},"tags":[],"trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}