{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport string\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nprint(os.listdir(\"../input\"))","execution_count":27,"outputs":[]},{"metadata":{"_uuid":"daaf38d20ae29a25605b94f0a1448ae2e85346f2"},"cell_type":"markdown","source":"Load data from csv"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"tr = pd.read_csv('../input/train.csv', parse_dates=['activation_date'])  # 1503424\nte = pd.read_csv('../input/test.csv',  parse_dates=['activation_date'])  # 508438","execution_count":2,"outputs":[]},{"metadata":{"_uuid":"5d9c5563b168121cac4e2237cdd563c42536c444"},"cell_type":"markdown","source":"Concat all of them for easliy do some feature engineering transeform."},{"metadata":{"trusted":true,"_uuid":"aa64e18f26a45a0872c002b049781c3d581db624"},"cell_type":"code","source":"daset = pd.concat([tr, te], axis=0)","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f750a37fd2538d7f5893b6f205cc62dd6148a5c"},"cell_type":"code","source":"punct = set(string.punctuation)\nprint(punct)","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"085df667cf4ce6395646c54af96570924bd38afa"},"cell_type":"code","source":"emoji = set()\nfor s in daset['title'].fillna('').astype(str):\n    for c in s:\n        if c.isdigit() or c.isalpha() or c.isalnum() or c.isspace() or c in punct:\n            continue\n        emoji.add(c)\n\nfor s in daset['description'].fillna('').astype(str):\n    for c in str(s):\n        if c.isdigit() or c.isalpha() or c.isalnum() or c.isspace() or c in punct:\n            continue\n        emoji.add(c)\n        \nprint(''.join(emoji))","execution_count":9,"outputs":[]},{"metadata":{"_uuid":"1a14c3f647b0182eac3b4327c182c3cff94fe851"},"cell_type":"markdown","source":"Oh Oh.. There are some many emojis in the textual dataset, please take care to deal with these.\n\nLater, I will give example for very sample features for each text columns. It is maybe help improve your score at LB."},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"c06dda718610ed953eb82b7c6df545f278d34116"},"cell_type":"code","source":"# basic word and char stats for title\ndaset['n_titl_len'] = daset['title'].fillna('').apply(len)\ndaset['n_titl_wds'] = daset['title'].fillna('').apply(lambda x: len(x.split(' ')))\ndaset['n_titl_dig'] = daset['title'].fillna('').apply(lambda x: sum(c.isdigit() for c in x))\ndaset['n_titl_cap'] = daset['title'].fillna('').apply(lambda x: sum(c.isupper() for c in x))\ndaset['n_titl_spa'] = daset['title'].fillna('').apply(lambda x: sum(c.isspace() for c in x))\ndaset['n_titl_pun'] = daset['title'].fillna('').apply(lambda x: sum(c in punct for c in x))\ndaset['n_titl_emo'] = daset['title'].fillna('').apply(lambda x: sum(c in emoji for c in x))\n\n# some ratio stats for title\ndaset['r_titl_wds'] = daset['n_titl_wds']/(daset['n_titl_len']+1)\ndaset['r_titl_dig'] = daset['n_titl_dig']/(daset['n_titl_len']+1)\ndaset['r_titl_cap'] = daset['n_titl_cap']/(daset['n_titl_len']+1)\ndaset['r_titl_spa'] = daset['n_titl_spa']/(daset['n_titl_len']+1)\ndaset['r_titl_pun'] = daset['n_titl_pun']/(daset['n_titl_len']+1)\ndaset['r_titl_emo'] = daset['n_titl_emo']/(daset['n_titl_len']+1)\n\n# basic word and char stats for description\ndaset['n_desc_len'] = daset['description'].fillna('').apply(len)\ndaset['n_desc_wds'] = daset['description'].fillna('').apply(lambda x: len(x.split(' ')))\ndaset['n_desc_dig'] = daset['description'].fillna('').apply(lambda x: sum(c in punct for c in x))\ndaset['n_desc_cap'] = daset['description'].fillna('').apply(lambda x: sum(c.isdigit() for c in x))\ndaset['n_desc_pun'] = daset['description'].fillna('').apply(lambda x: sum(c.isupper() for c in x))\ndaset['n_desc_spa'] = daset['description'].fillna('').apply(lambda x: sum(c.isspace() for c in x))\ndaset['n_desc_emo'] = daset['description'].fillna('').apply(lambda x: sum(c in emoji for c in x))\ndaset['n_desc_row'] = daset['description'].astype(str).apply(lambda x: x.count('/\\n'))\n\n# some ratio stats\ndaset['r_desc_wds'] = daset['n_desc_wds']/(daset['n_desc_len']+1)\ndaset['r_desc_dig'] = daset['n_desc_dig']/(daset['n_desc_len']+1)\ndaset['r_desc_cap'] = daset['n_desc_cap']/(daset['n_desc_len']+1)\ndaset['r_desc_spa'] = daset['n_desc_spa']/(daset['n_desc_len']+1)\ndaset['r_desc_pun'] = daset['n_desc_pun']/(daset['n_desc_len']+1)\ndaset['r_desc_row'] = daset['n_desc_row']/(daset['n_desc_len']+1)\ndaset['r_desc_emo'] = daset['n_desc_emo']/(daset['n_desc_len']+1)\n\ndaset['r_titl_des'] = daset['n_titl_len']/(daset['n_desc_len']+1)","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"fcb98036312019ce7dcb9d03bc218477466afd88"},"cell_type":"markdown","source":"How to measure the feature quality, let's plot the corr"},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f12aa21de03725cb796d4049fb64205b8149d36a"},"cell_type":"code","source":"text_feature = list(daset.filter(regex='r_titl|n_titl|r_desc|n_desc').columns)\ndata = daset.loc[~daset['deal_probability'].isnull(), text_feature+['deal_probability']]\ndata.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"50645d5bc840bda6bae476e42a131120097347de"},"cell_type":"code","source":"def plot_corr(corr,method):\n    # Generate a mask for the upper triangle\n    mask = np.zeros_like(corr, dtype=np.bool)\n    mask[np.triu_indices_from(mask)] = True\n\n    # Set up the matplotlib figure\n    f, ax = plt.subplots(figsize=(16, 13))\n\n    # Generate a custom diverging colormap\n    cmap = sns.diverging_palette(220, 10, as_cmap=True)\n\n    # Draw the heatmap with the mask and correct aspect ratio\n    sns.heatmap(corr, mask=mask, cmap=cmap, vmax=.3, center=0, square=True, linewidths=.5, cbar_kws={\"shrink\": .5})\n    plt.savefig('./corr_{}.jpg'.format(method))","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"b5f3eb8b3accbdda8c8c12fc517d025c661fb281"},"cell_type":"code","source":"corr = data.corr(method='pearson')\nplot_corr(corr, 'pearson')","execution_count":23,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d900f798bc1eecec4367df3587e161b2d12e609"},"cell_type":"code","source":"","execution_count":28,"outputs":[]},{"metadata":{"_uuid":"c766ec5cdf78d8d1014d41b61e24d7f16048dda8"},"cell_type":"markdown","source":"In the end, I want to tell you these above features really good for me, hope to help you. Thanks for your time reading."}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}