{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"![](https://i.ytimg.com/vi/Dd_NgYVOdLk/maxresdefault.jpg)youtube.com","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":false,"_kg_hide-output":false}},{"cell_type":"markdown","source":"\"Predictions in phase1 are evaluated by Levenshtein Mean Distance between the predicted sentence and the ground truth sentence. The Levenshtein distance between two sentences is the minimum number of single-character edits (insertions, deletions, or substitutions) required to change one sentence into the other\"\n\nhttps://www.kaggle.com/competitions/dlsprint/overview/evaluation","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:17:26.561111Z","iopub.execute_input":"2022-08-13T00:17:26.562430Z","iopub.status.idle":"2022-08-13T00:17:26.591048Z","shell.execute_reply.started":"2022-08-13T00:17:26.562321Z","shell.execute_reply":"2022-08-13T00:17:26.589898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/dlsprint/train.csv\", delimiter=',', encoding='utf8')\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:27:11.773481Z","iopub.execute_input":"2022-08-13T00:27:11.773911Z","iopub.status.idle":"2022-08-13T00:27:13.077756Z","shell.execute_reply.started":"2022-08-13T00:27:11.773878Z","shell.execute_reply":"2022-08-13T00:27:13.076589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val = pd.read_csv(\"/kaggle/input/dlsprint/validation.csv\", delimiter=',', encoding='utf8')\nval.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:37:57.429731Z","iopub.execute_input":"2022-08-13T00:37:57.430089Z","iopub.status.idle":"2022-08-13T00:37:57.532797Z","shell.execute_reply.started":"2022-08-13T00:37:57.430059Z","shell.execute_reply":"2022-08-13T00:37:57.531658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#By Daniel Legorreta","metadata":{}},{"cell_type":"code","source":" def Levenshtein(s0, s1):\n        if s0 is None:\n            raise TypeError(\"Argument s0 is NoneType.\")\n        if s1 is None:\n            raise TypeError(\"Argument s1 is NoneType.\")\n        if s0 == s1:\n            return 0.0\n        if len(s0) == 0:\n            return len(s1)\n        if len(s1) == 0:\n            return len(s0)\n\n        v0 = [0] * (len(s1) + 1)\n        v1 = [0] * (len(s1) + 1)\n\n        for i in range(len(v0)):\n            v0[i] = i\n\n        for i in range(len(s0)):\n            v1[0] = i + 1\n            for j in range(len(s1)):\n                cost = 1\n                if s0[i] == s1[j]:\n                    cost = 0\n                v1[j + 1] = min(v1[j] + 1, v0[j + 1] + 1, v0[j] + cost)\n            v0, v1 = v1, v0\n\n        return v0[len(s1)]\n\ndef distance(s0,s1):\n    if s0 == s1:\n            return 0.0\n        \n    m_len = max(len(s0), len(s1))\n    if m_len == 0:\n        return 0.0\n     \n    return Levenshtein(s0, s1) / m_len\n\ndef similarity(s0, s1):\n        return 1.0 - distance(s0, s1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:18:45.081542Z","iopub.execute_input":"2022-08-13T00:18:45.082084Z","iopub.status.idle":"2022-08-13T00:18:45.095664Z","shell.execute_reply.started":"2022-08-13T00:18:45.082035Z","shell.execute_reply":"2022-08-13T00:18:45.094046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Leveshtein_Metric(x):\n    \"aux funct\"\n    a=similarity(x[''],x['target'])#We don't have target. I saved it for the next time.\n    return a","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#By Meghan Jambhale","metadata":{}},{"cell_type":"code","source":"fails = []\n\ndef return_default_value_if_fails(default_value):\n\n    def decorator(func):\n        def inner(*args, **kwargs):\n            try:\n                return func(*args, **kwargs)\n            except Exception as e:\n                fails.append((func, (args, kwargs), e))\n                return default_value\n        return inner\n\n    return decorator","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:43:35.729734Z","iopub.execute_input":"2022-08-13T00:43:35.730189Z","iopub.status.idle":"2022-08-13T00:43:35.737253Z","shell.execute_reply.started":"2022-08-13T00:43:35.730155Z","shell.execute_reply":"2022-08-13T00:43:35.736117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def getResults(questions, fn):\n    @return_default_value_if_fails(default_value=0.1)\n    def getResult(q):\n        answer, up_votes, sentence = fn(q)\n        return [path, sentence, accents, up_votes]\n    output=pd.DataFrame(list(map(getResult, questions)), columns=[\"path\", \"sentence\", \"accents\", \"up_votes\"])\n    return output\ntrain_data=df[\"path\"].tolist()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:43:40.495029Z","iopub.execute_input":"2022-08-13T00:43:40.495843Z","iopub.status.idle":"2022-08-13T00:43:40.509487Z","shell.execute_reply.started":"2022-08-13T00:43:40.495803Z","shell.execute_reply":"2022-08-13T00:43:40.508024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from Levenshtein import ratio\ndata=df\ndef gettingApproximateAnswer(q):\n    max_score = 0\n    answer = \"\"\n    prediction = \"\"\n    for idx, row in data.iterrows():\n        score = ratio(row[\"path\"], q)\n        if score >= 0.9: #You can stop\n            return row[\"sentence\"], score, row[\"sentence\"]\n        elif score > max_score: # Need to continue because unsure\n            max_score = score\n            answer = row[\"sentence\"]\n            prediction = row[\"sentence\"]\n    if max_score > 0.3:\n        return answer, max_score, prediction\n    \n    return \"Apology.I couldn't get you.\", max_score, prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:36:24.406866Z","iopub.execute_input":"2022-08-13T00:36:24.407323Z","iopub.status.idle":"2022-08-13T00:36:24.414639Z","shell.execute_reply.started":"2022-08-13T00:36:24.407286Z","shell.execute_reply":"2022-08-13T00:36:24.413746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_data=val['path'].tolist()\noutput=getResults(val_data, gettingApproximateAnswer)\noutput","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:43:57.473721Z","iopub.execute_input":"2022-08-13T00:43:57.474203Z","iopub.status.idle":"2022-08-13T00:56:51.983154Z","shell.execute_reply.started":"2022-08-13T00:43:57.474159Z","shell.execute_reply":"2022-08-13T00:56:51.981350Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Above ValueError: Shape of passed values is (7747, 1), indices imply (7747, 4)","metadata":{}},{"cell_type":"markdown","source":"#By Nitin Singh","metadata":{}},{"cell_type":"code","source":"str1 = 'বাবা সত্যেন ঘোষ।'\nstr2 = 'আপনি খুব একটা কথা বলার লোক নন, তাই না?'","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:27:51.339551Z","iopub.execute_input":"2022-08-13T00:27:51.340054Z","iopub.status.idle":"2022-08-13T00:27:51.346201Z","shell.execute_reply.started":"2022-08-13T00:27:51.339979Z","shell.execute_reply":"2022-08-13T00:27:51.345057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr = [[None for _ in range(len(str1)+1)] for _ in range(len(str2)+1)]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:28:12.048260Z","iopub.execute_input":"2022-08-13T00:28:12.049925Z","iopub.status.idle":"2022-08-13T00:28:12.056934Z","shell.execute_reply.started":"2022-08-13T00:28:12.049862Z","shell.execute_reply":"2022-08-13T00:28:12.055365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(str2)+1):\n    arr[i][0] = i\n\nfor j in range(1, len(str1)+1):\n    arr[0][j] = j\n    arr[1][j] = j-1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:28:29.740531Z","iopub.execute_input":"2022-08-13T00:28:29.740989Z","iopub.status.idle":"2022-08-13T00:28:29.748226Z","shell.execute_reply.started":"2022-08-13T00:28:29.740949Z","shell.execute_reply":"2022-08-13T00:28:29.746854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:28:46.549295Z","iopub.execute_input":"2022-08-13T00:28:46.549772Z","iopub.status.idle":"2022-08-13T00:28:46.570415Z","shell.execute_reply.started":"2022-08-13T00:28:46.549731Z","shell.execute_reply":"2022-08-13T00:28:46.569252Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(2, len(str2)+1):\n    for j in range(1, len(str1)+1):\n\n        if str1[j-1] == str2[i-1]:\n            arr[i][j] = arr[i-1][j-1]\n\n        else:\n            arr[i][j] = min(arr[i-1][j], arr[i-1][j-1], arr[i][j-1]) + 1","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:29:14.639783Z","iopub.execute_input":"2022-08-13T00:29:14.640160Z","iopub.status.idle":"2022-08-13T00:29:14.647611Z","shell.execute_reply.started":"2022-08-13T00:29:14.640128Z","shell.execute_reply":"2022-08-13T00:29:14.646731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arr","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:29:30.156607Z","iopub.execute_input":"2022-08-13T00:29:30.157104Z","iopub.status.idle":"2022-08-13T00:29:30.179039Z","shell.execute_reply.started":"2022-08-13T00:29:30.157008Z","shell.execute_reply":"2022-08-13T00:29:30.177737Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"levenshtein_dist = arr[-1][-1]","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:29:56.899311Z","iopub.execute_input":"2022-08-13T00:29:56.899724Z","iopub.status.idle":"2022-08-13T00:29:56.904135Z","shell.execute_reply.started":"2022-08-13T00:29:56.899688Z","shell.execute_reply":"2022-08-13T00:29:56.903358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"levenshtein_dist","metadata":{"execution":{"iopub.status.busy":"2022-08-13T00:30:12.026001Z","iopub.execute_input":"2022-08-13T00:30:12.026417Z","iopub.status.idle":"2022-08-13T00:30:12.032888Z","shell.execute_reply.started":"2022-08-13T00:30:12.026382Z","shell.execute_reply":"2022-08-13T00:30:12.031997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"![](https://image.slidesharecdn.com/201711grdfdataquality-171120112851/95/how-data-science-can-help-energy-companies-map-their-infrastructure-7-638.jpg?cb=1511177590)slideshare.net","metadata":{}},{"cell_type":"markdown","source":"#Acknowledgements:\n\nDaniel Legorreta https://www.kaggle.com/code/legorreta/traditional-approach-levenshtein\n\nMeghal Jambhale https://www.kaggle.com/code/meghaljambhale/levenshtein-distance-to-predict-correct-aswer\n\nNitin Singh https://www.kaggle.com/code/nitinsss/levenshtein-distance-dynamic-programming\n\nhttps://youtu.be/We3YDTzNXEk","metadata":{}}]}