{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nDATA_DIR = '../input/'\nos.listdir(DATA_DIR)","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"e2c4bbef09e11f23526222696b6810495e8ba80c"},"cell_type":"code","source":"import time\nfrom contextlib import contextmanager\nfrom functools import lru_cache\nos.environ['OMP_NUM_THREADS'] = '4'\n\n@contextmanager\ndef timer(name):\n    t0 = time.time()\n    yield\n    print(f'[{name}] done in {time.time() - t0:.1f} s')","execution_count":9,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cc6afef39398ab22c671b90b7211b418c80b638","collapsed":true},"cell_type":"code","source":"df_names = ['train.csv', \n            'train_active.csv', \n            'test.csv', \n            'test_active.csv'\n            ]\n\ndef get_group_feat(by, y='price', mode='ratio', df_names=df_names):\n    print('Working on', by)\n    if not isinstance(by, list):\n        by = [by]\n    usecols = by+[y]\n    if 'image_top_1' in by or 'image' in by:\n        df_names = ['train.csv', 'test.csv']\n    df = []\n    for fname in df_names:\n        print('loading', fname)\n        df.append(pd.read_csv(DATA_DIR+fname, usecols=usecols))\n        if y=='price':\n            df[-1][y] = np.log1p(df[-1][y].fillna(0)).astype('float32')\n        if fname in ['train.csv', 'test.csv']:\n            df[-1]['eval'] = 1\n        else:\n            df[-1]['eval'] = 0\n        df[-1]['eval'] = df[-1]['eval'].astype('uint8')\n        if 'image' in by:\n            df[-1]['image'] = (~df[-1]['image'].isnull()).astype('uint8')\n    with timer('concating'):\n        df = pd.concat(df, ignore_index=True)\n    gc.collect();\n    if len(by)>1:\n        p_colname = 'P_'+'X'.join(by)\n    else:\n        p_colname = by[0]\n    grp = df.groupby(by, sort=False)[y]\n    if mode=='rank':\n        with timer('ranking'):\n            feat = grp.rank(na_option='top', ascending=True, pct=True).to_frame()\n    elif mode=='ratio':\n        with timer('calculating ratio'):\n            feat = grp.apply(lambda x: (x+1e-12)/(x.max()+1e-12))\n    elif mode=='zscore':\n        with timer('calculating zscore'):\n            zscore = lambda x: (x - x.mean()) / x.std()\n            feat = grp.apply(zscore)\n    feat.columns = [p_colname+'_prk']\n    feat = feat[df['eval']==1].reset_index(drop=True)\n    featname = p_colname+'_prk'\n    gc.collect();\n    return feat, featname","execution_count":10,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"94f71cc0617d2c7fffe2a189dc663c8947f1c2ad"},"cell_type":"code","source":"by_li = [#'item_id',\n         'user_id',\n         'region', \n         'city', \n         'parent_category_name', \n         'category_name',\n         'param_1', \n         'param_2', \n         'param_3',\n         #'title',\n         #'description',\n         #'price',\n         #'item_seq_number', \n         'activation_date',\n         'user_type',\n         'image',\n         'image_top_1',\n         #'deal_probability',\n        ]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bf334672901376cdac241637f62b34c42f8c80c0","_kg_hide-output":true,"scrolled":false},"cell_type":"code","source":"df = pd.DataFrame()\nfor idx, by in enumerate(by_li):\n    gc.collect();\n    print('No.', idx+1, 'out of', len(by_li))\n    feat, name = get_group_feat(by)\n    df[by] = feat.values\ndf = df[by_li]\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4282aecbd5fab564838c17b2dc3e4e67a012eeb7"},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"6dfcc8a219370cb9ee72df44348b7c5bbd2b1298"},"cell_type":"code","source":"df.to_csv('price_ratio_enc.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}