{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport spacy\nimport networkx as nx\nfrom itertools import combinations\nfrom collections import defaultdict\nfrom collections import Counter\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"data = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\", nrows=1000, encoding=\"ISO-8859-1\")\n#data = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ndata.head()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#common entities list would be like the one that we are getting it \n#edge of co-ccurance\n\ndef cooccurrence(common_entities):\n    com = defaultdict(int)\n    #Build co-occurence matrix\n    for w1,w2 in combinations(sorted(common_entities),2):\n        com[w1,w2] += 1\n    result = defaultdict(dict)\n    for (w1,w2), count in com.items():\n        if w1 != w2:\n            #result[w1][w2]={'weight':count}\n            result[w1][w2]= count\n    return result\n\ndef combine_nested_dicts(x, y):\n    all_keys = set().union(x.keys(), y.keys())\n    print(all_keys)\n    for key in all_keys:\n        if key in y and key in x :\n            x[key] = Counter(x[key]) + Counter(y[key])\n            x[key] = dict(x[key])\n            print(x[key])\n        else:\n            x.update(y[key])\n    for key in y:\n        if key not in x:\n            x.update(y[key])\n    return (x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# test coocurrence \nprint(cooccurrence('abcaqwvv'))\n\n# test combining dicts\nx = cooccurrence('abb')\ny = cooccurrence('acbd')\nz = cooccurrence('efg')\nprint(combine_nested_dicts(x, z))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"nlp = spacy.load('en_core_web_sm')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"doc = nlp(u'Apple is looking at buying U.K. startup for $1 billion')\nfor ent in doc.ents:\n    print(ent.text, ent.start_char, ent.end_char, ent.label_)\nnlp = spacy.load('en_core_web_sm') #dictionary of words that contains english words conent from \ndocs = nlp.pipe(data)\n#cooccurrence(docs.ents)\n#initialize claim counter and \nclaim_counter = 0\nlist_indexes = []\nlist_entities = []\nfor doc in docs:\n    #print(doc)\n    for ent in doc.ents:\n        print (ent)\n        list_indexes.append(claim_counter)\n        list_entities.append(ent.text)\n        print(ent.text)\n        print(ent.start_char)\n        print(ent.end_char)\n        print(ent.label_)\n    claim_counter +=1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import h2o\nh2o.init()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data=h2o.import_file(\"/input/glove480small-rao/train_glove480_small.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"","_uuid":"","trusted":true},"cell_type":"code","source":"import pandas as pd\ntrain_paragram_small = pd.read_csv(\"../input/train_paragram_small.csv\")","execution_count":0,"outputs":[]},{"metadata":{"_cell_guid":"","_uuid":"","trusted":true},"cell_type":"code","source":"import pandas as pd\ntrain_google_news_small = pd.read_csv(\"../input/train_google_news_small.csv\")","execution_count":0,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":1}