{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nfrom plotly.offline import init_notebook_mode, iplot\nfrom plotly import offline as ply\nfrom wordcloud import WordCloud\nimport seaborn as sns\nimport plotly.graph_objs as go\nimport plotly.plotly as py\nfrom plotly import tools\nfrom matplotlib_venn import venn2, venn2_circles\nimport string\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nimport nltk\nfrom nltk.corpus import stopwords\nimport scipy\nimport os\nimport lightgbm as lgb\n\ninit_notebook_mode(connected=True)\nsns.set()\n\nprint(os.listdir(\"../input\"))\n","execution_count":2,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"%%time\n# Load Data\ntrain_df = pd.read_csv('../input/train.csv', index_col = \"item_id\", parse_dates = [\"activation_date\"])\ntest_df = pd.read_csv('../input/test.csv', index_col=\"item_id\", parse_dates = [\"activation_date\"])\n\n# Shape\nprint(train_df.shape)\nprint(test_df.shape)","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cbeaef100fb26d751e7942749f0beaaaa0bd036"},"cell_type":"code","source":"# Label\nlabel = train_df.deal_probability.copy()\ndf = train_df.drop('deal_probability', axis=1)\n\n# Combine Train & Test\ndf = pd.concat([df, test_df],axis=0)\n\ndf.head(5)","execution_count":9,"outputs":[]},{"metadata":{"_uuid":"27752d4b61e9f3ce58fc1ba4f5110f84129a8dd4"},"cell_type":"markdown","source":"### User ID\n\n1. Overlap between `user_id` in train and test"},{"metadata":{"trusted":true,"_uuid":"d3edcf73f889febdb7a50c24de1c5e07eef90179","collapsed":true},"cell_type":"code","source":"# User Count\nprint(\"All User:\", len(set(df.user_id)))\nprint(\"User in TrainSet:\", len(set(train_df.user_id)))\nprint(\"User in TestSet:\", len(set(test_df.user_id)))\n\nprint(\"User in TrainSet & User in TestSet\", len(set(train_df.user_id)&set(test_df.user_id)))\n\n\n# Venn\n\nfig, axarr = plt.subplots(1, 1, figsize=(14, 7))\n\ndef get_venn(axarr, feature):\n    axarr.set_title(f'Overlap between {feature} in train and test')\n    venn2([\n        set(train_df[feature].values), \n        set(test_df[feature].values)\n    ], set_labels = ('train', 'test'), ax=axarr)\nget_venn(axarr, 'user_id')\n","execution_count":8,"outputs":[]},{"metadata":{"_uuid":"38a2433761ef20af5600116f434f81b7017bf011"},"cell_type":"markdown","source":"### Activation Date\n"},{"metadata":{"trusted":true,"_uuid":"96901954de26fd14602bd8050691bd7ddabcd982","collapsed":true},"cell_type":"code","source":"print(f\"All Date from {df.activation_date.min()} to {df.activation_date.max()}.\")\nprint(f\"Train Date from {train_df.activation_date.min()} to {train_df.activation_date.max()}.\")\nprint(f\"Test Date from {test_df.activation_date.min()} to {test_df.activation_date.max()}.\")","execution_count":9,"outputs":[]},{"metadata":{"_uuid":"a9970affa1a5cfe3e86f527abb28be0c594fde6a"},"cell_type":"markdown","source":"Train_df & Test_df happend in two different periods.\n\nSo features `[Weekd of Year, Day of Month]` in many kernels is useless."},{"metadata":{"_uuid":"c76e6187453318a570d881ad339f049306068c85"},"cell_type":"markdown","source":"### User Type"},{"metadata":{"trusted":true,"_uuid":"222668eec0332aa02626d0d2a81297e936820cc2","collapsed":true},"cell_type":"code","source":"# 3 types\nprint('All Types: ', set(df.user_type))\n# Distribution Of User Type\ndef _generate_bar_plot_ver(df, col, title, color, w=None, h=None, lm=0, limit=100, need_trace = False):\n    cnt_srs = df[col].value_counts()[:limit]\n    trace = go.Bar(x=list(cnt_srs.index), y=list(cnt_srs.values),\n        marker=dict(color = color))\n    if need_trace:\n        return trace\n    if w != None and h != None:\n        layout = dict(title=title, margin=dict(l=lm), width=w, height=h)\n    else:\n        layout = dict(title=title, margin=dict(l=lm))\n    data = [trace]\n    fig = go.Figure(data=data, layout=layout)\n    iplot(fig)\n\ndef distribute(cols):\n    trace1 = _generate_bar_plot_ver(df, cols, \"All\", ['#f25771','#93ef51'], lm=0, limit=30, need_trace = True)\n    trace2 = _generate_bar_plot_ver(train_df, cols, \"Train\", ['#f25771','#93ef51'], 200, limit=30, need_trace = True)\n    trace3 = _generate_bar_plot_ver(test_df, cols, \"Train\", ['#f25771','#93ef51'], 200, limit=30, need_trace = True)\n\n    fig = tools.make_subplots(rows=1, cols=3, specs=[[{'colspan': 1}, {},{}]], print_grid=False, subplot_titles = ['All','Train','Test'])\n    fig.append_trace(trace1, 1, 1);\n    fig.append_trace(trace2, 1, 2);\n    fig.append_trace(trace3, 1, 3);\n\n    fig['layout'].update(height=400, title='',showlegend=False)\n    iplot(fig)\n\ndistribute('user_type')","execution_count":18,"outputs":[]},{"metadata":{"_uuid":"96af4129a7378bca29f63ba63130af116424a44b"},"cell_type":"markdown","source":"### City"},{"metadata":{"trusted":true,"_uuid":"448dc08f8197f1c0e60c7be2ecf75970aa0981e4","collapsed":true},"cell_type":"code","source":"distribute('city')\n\nprint(\"All City:\", len(set(df.city)))\nprint(\"City in TrainSet:\", len(set(train_df.city)))\nprint(\"City in TestSet:\", len(set(test_df.city)))\n\nprint(\"City in TrainSet & City in TestSet\", len(set(train_df.city)&set(test_df.city)))\n\n# There are 19 citys not in TrainSet","execution_count":22,"outputs":[]},{"metadata":{"_uuid":"35735e86e70baff41ce463e701579127bb4be745"},"cell_type":"markdown","source":"### Image"},{"metadata":{"trusted":true,"_uuid":"b95689690834ba5c137e3d609a7981bba80b25f7","collapsed":true},"cell_type":"code","source":"df.loc['b912c3c6a6ad', 'image']","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}