{"cells":[{"metadata":{"collapsed":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":false},"cell_type":"code","source":"# This notebook explores the character distribution of description.\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport unicodedata\nfrom sklearn.feature_extraction.text import CountVectorizer\nfrom matplotlib import pyplot","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"639dc4e8b8b3a679ccd17152c71c7e0dcaef48e1","_cell_guid":"11419193-6309-450e-b0b7-68c2a3386e51"},"cell_type":"markdown","source":"Let's look at the character distribution of the description by counting all the characters with a CountVectorizer."},{"metadata":{"collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":false},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', usecols=['description', 'deal_probability'])\ntest = pd.read_csv('../input/test.csv', usecols=['description'])\n\ndf = pd.concat((train, test))\n\ndf.index = range(df.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"fc142201da82defd732dfa9db8bf97104558d1ad","_cell_guid":"13ea7814-f3cc-44b1-a09d-c98015711a18","trusted":false},"cell_type":"code","source":"charvec = CountVectorizer(\n    analyzer='char',\n    lowercase=False,\n    max_df=1.0,\n    min_df=1\n)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"f653474035d7c4b13d7718e2ff830146b16a1302","_cell_guid":"4dfa4223-bd1a-4e53-941d-49f2ba40dfef","trusted":false},"cell_type":"code","source":"char_counts = charvec.fit_transform(df['description'].fillna(''))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"453ea94370c8a6439f56246501635bd89388303c","_cell_guid":"82bd92e7-40c7-4d05-a2d4-f04676475d54"},"cell_type":"markdown","source":"We got 1749 different characters. Neat!"},{"metadata":{"collapsed":true,"_uuid":"2b88e36795dcefae4e6980eafa09d645acc543e8","_cell_guid":"35c11e1d-690e-453c-91c1-f74cd3ffed02","trusted":false},"cell_type":"code","source":"char_counts","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ddec3f662e6f13f1f469446bfa54386664ce7714","_cell_guid":"396f8fc5-6e81-4ab5-acf5-96f6b1ff9a5f"},"cell_type":"markdown","source":"We can add them all together to get the total character distribution from all the descriptions."},{"metadata":{"collapsed":true,"_uuid":"c256573f84f5afa298ba4847d89073f889f35a2b","_cell_guid":"d720fae6-dfd1-43d9-86ae-b3ff6f0073f3","trusted":false},"cell_type":"code","source":"totals = pd.DataFrame(\n    np.array(char_counts.sum(axis=0))[0], \n    index=charvec.get_feature_names(),\n    columns=['cnt']\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f7b36c1447f779065ddb6c01c087529fca0b60fc","_cell_guid":"557b005a-3257-4275-b16c-764b8e8af68b"},"cell_type":"markdown","source":"We can also use unicodedata to capture some meta information of the caaracters."},{"metadata":{"collapsed":true,"_uuid":"c643bcf6fdd0ceba423614baf735e8b7ff9198f4","_cell_guid":"3525bd7a-9565-4052-8151-d1369dcbe2ed","trusted":false},"cell_type":"code","source":"totals['ord'] = totals.index.map(lambda x: ord(x))\ntotals['cat'] = totals.index.map(lambda x: unicodedata.category(x))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"6ac446251640813f3da8c1ac2636d395e44bf9fd","_cell_guid":"17d8a25c-4ad9-4aa1-a59e-c3eb5dadd6b2","trusted":false},"cell_type":"code","source":"def extract_name(x):\n    try:\n        if '\\t' == x:\n            return 'CHARACTER TABULATION'\n        if '\\n' == x:\n            return 'LINE FEED'\n        return unicodedata.name(x)\n    except:\n        return None\n    \ntotals['name'] = totals.index.map(extract_name)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"08de94c759b515bc15eb77c2ee967ace28239247","_cell_guid":"3068b47a-4012-4c86-880a-b3a77a33addb","trusted":false},"cell_type":"code","source":"totals['name'].fillna('', inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"53159af3593ea25299523fe50194ddbecb3dd3f7","_cell_guid":"c3afb0b9-619f-466f-a3f6-976084e656f9"},"cell_type":"markdown","source":"The two-char codes in *cat* stand for different character sets. For example Ll stands for lowercase letter, Zs stands for space and separator. You can check each category them out [here](https://en.wikipedia.org/wiki/Unicode_character_property)."},{"metadata":{"collapsed":true,"_uuid":"2eec539761d96a82f8be0f0a64233234231c066b","_cell_guid":"13e45109-9c2a-4b3a-b1e1-154b6789e206","trusted":false},"cell_type":"code","source":"r = totals.groupby('cat').cnt.agg(['count', 'sum'])\nr.sort_values('sum', ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5ea687bb602d13c937b7340c26e4a56e083e85d0","_cell_guid":"6053e02b-d888-49d3-90d8-4318ff78f896"},"cell_type":"markdown","source":"The majority is lower case letters, spaces, punctuation and upper case letters. There are also some numeric characters."},{"metadata":{"collapsed":true,"_uuid":"a81db0ff35758b2e470b22518f7fd5c3abe272ce","_cell_guid":"1406c934-c927-4ecb-9e8b-46d1b3b277f7","trusted":false},"cell_type":"code","source":"(r / r.sum()).sort_values('sum', ascending=False).plot(kind='bar', figsize=(12, 4))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a6e88fb98d4ace1cfd0c823b29d7767f5c1fc957","_cell_guid":"5cc910f8-1d88-4e20-8d09-bdd36bb61410"},"cell_type":"markdown","source":"Let's look at the mean of deal_probability for different character set counts. For eg. punctuation."},{"metadata":{"collapsed":true,"_uuid":"61a3edbfa31696afd41c11f492ea7c86181cce14","_cell_guid":"983ee143-0edc-4822-8dbe-a4e712121daf","trusted":false},"cell_type":"code","source":"charset_idx = np.array(range(totals.shape[0]))[totals['cat'] == 'Po']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7521f191c5b4d85aaecc0d0f8d7dd9149ac86ac2","_cell_guid":"f43f2564-ea30-4fb3-97c3-b7b7a4b0c902"},"cell_type":"markdown","source":"Let's bin the input to draw a neato chart"},{"metadata":{"collapsed":true,"_uuid":"a625269c381ea1202c948a198a7ef5e841c655c9","_cell_guid":"abd16d86-ca3f-45f3-8a93-747b4369d8c9","trusted":false},"cell_type":"code","source":"df['charset_cnt'] = np.log2(char_counts[:, charset_idx].sum(axis=1) + 1).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"bb0d00cb74051ab65f12ffa5deae820b2dbb55e0","_cell_guid":"03a3d49f-7616-4206-b090-9295cdea0cea","trusted":false},"cell_type":"code","source":"df.groupby('charset_cnt').deal_probability.mean().plot(kind='bar', color='#7777ac')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"a5f781069b55d32e7916b28f7d0d82eaeecd2351","_cell_guid":"619e62e3-03da-4496-b721-59785bef13b7","trusted":false},"cell_type":"code","source":"del df['charset_cnt']","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f6381c1707227da26e4e9018fddd95ef7b7c7497","_cell_guid":"9c841b44-ea2d-49c3-a0e2-9bf218bb4799"},"cell_type":"markdown","source":"We can do this for all character sets eventually."},{"metadata":{"collapsed":true,"_uuid":"f432d5f618bd0aa00a0d0a1676098de1a5fc4be8","_cell_guid":"22979f0f-1720-4a04-8587-17775342f928","trusted":false},"cell_type":"code","source":"for cat in totals['cat'].unique():\n    print(cat)\n    feature = 'charset_{}_cnt'.format(cat)\n    charset_idx = np.array(range(totals.shape[0]))[totals['cat'] == cat]\n    df[feature] = np.log2(char_counts[:, charset_idx].sum(axis=1) + 1).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"907b6a82cf6ea052c30d37f25d398fc3003e1d90","_cell_guid":"8261827a-544d-4028-86db-9a1c9b6d67df","trusted":false},"cell_type":"code","source":"nu_cats = totals.cat.nunique()","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"c2ce4137c04908811ee81da48edda4ef4901ec5f","_cell_guid":"4b33be35-232b-4ac2-a8f2-7e9114a660d3","trusted":false},"cell_type":"code","source":"nu_cats","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"9a8015351540615bd83182acfb28e2242fc52c09","_cell_guid":"91340d62-4f85-443c-9319-cbc1bf59acea","trusted":false},"cell_type":"code","source":"charset_cols = list(filter(lambda x: x.startswith('charset_'), df.columns))","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"cae5ab08ecf28ffb9f50b3ef0bf56f4274887792","_cell_guid":"91cc6c5d-a9af-42f1-ae89-70756ddb94cb","trusted":false},"cell_type":"code","source":"max_vals = df[charset_cols].max().sort_values(ascending=False)","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"4449f918baca21ea6d60ac02bef0a4d3e384624c","_cell_guid":"5a603534-3c99-4522-b4f1-b8413ebb92ef","trusted":false},"cell_type":"code","source":"max_vals[max_vals <= 8].index.shape","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"87dd6219abfd10683b15a829b21763f190cfb383","_cell_guid":"b415f4e2-61c3-4fa7-9643-2617d2f52a6d","trusted":false},"cell_type":"code","source":"f, axes = pyplot.subplots(3, 3, sharey=True, figsize=(15, 10))\naxes = axes.flatten()\nfor k, feat in enumerate(max_vals[max_vals > 8].index):\n    r = df.groupby(feat).deal_probability.agg(['count', 'mean'])\n    r['pcnt_cnt'] = r['count'] / r['count'].sum() \n    r[['pcnt_cnt', 'mean']].plot(kind='bar', color=['#667799', '#aa3366'], ax=axes[k], title=feat)\n    axes[k].set_xlabel('')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"52e8345ef087383a474877a6217eb9e76763a846","_cell_guid":"b1d157ce-b901-4ee5-b941-4facfcda71d3","trusted":false},"cell_type":"code","source":"f, axes = pyplot.subplots(3, 6, sharey=True, figsize=(15, 10))\naxes = axes.flatten()\nfor k, feat in enumerate(max_vals[max_vals <= 8].index):\n    r = df.groupby(feat).deal_probability.agg(['count', 'mean'])\n    r['pcnt_cnt'] = r['count'] / r['count'].sum() \n    r[['pcnt_cnt', 'mean']].plot(kind='bar', color=['#667799', '#aa3366'], ax=axes[k], title=feat)\n    axes[k].set_xlabel('')","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"3826d7f6e1baa89f1c1fdbe633336d69c8364082","_cell_guid":"e18987db-e3ca-46b2-b663-811b8cdd18b8","trusted":false},"cell_type":"code","source":"totals['idx'] = range(totals.shape[0])","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"ff13cd36a60ab27ce2926db8ff9094f75a7a3856","_cell_guid":"c54e7882-a469-49fa-bf50-5f8e5299fd8c","trusted":false},"cell_type":"code","source":"totals.loc['!']","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"3d597a069a5372f28a08b496e287e67bd1e1a214","_cell_guid":"a2b2bf05-517f-4910-a712-9dac20eae6f5","trusted":false},"cell_type":"code","source":"df['!_cnt'] = char_counts[:, 3]","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"27268eeee5d23bf45177991d4e7e3786f5906677","_cell_guid":"261e44bc-389b-4bdd-b683-4af9efe49e70","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"_uuid":"6b9994e1788392576851d53f8578a91b6425f69a","_cell_guid":"6c00606d-f4af-43ba-87aa-ac51e8b11d5b","trusted":false},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}