{"cells":[{"metadata":{"trusted":true,"collapsed":true,"_uuid":"c72f0396cfc2d43d054f7c12ae3fad5960fc1c27"},"cell_type":"markdown","source":"In this notebook, I'll try that fill missing values of price \nReading this kernel(https://www.kaggle.com/tunguz/bow-meta-text-and-dense-features-lb-0-2241),\nthe importance of price was high,so i think that fill missing values of price is very importance!"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"collapsed":true},"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n","execution_count":1,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"collapsed":true},"cell_type":"code","source":"#Data Load\ntrain =pd.read_csv(\"../input/train.csv\",index_col = \"item_id\", parse_dates = [\"activation_date\"])\n\ntest =pd.read_csv(\"../input/test.csv\",index_col = \"item_id\", parse_dates = [\"activation_date\"])\n\ndel(train[\"deal_probability\"])\ndf = pd.concat([train,test],axis = 0)","execution_count":2,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f21998b6f41ab44cd83e603e057f35e709d4b883","collapsed":true},"cell_type":"code","source":"print(len(df[\"price\"]))\nprint(df[\"price\"].isnull().sum())","execution_count":3,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6f3c6f1daaaf592d8fa33fff8b2e1838a3c32621","collapsed":true},"cell_type":"code","source":"def fillna(lis,df,target):\n    count = 0\n    while 0 < len(lis):\n        count += 1\n        colname = str(count)\n        print(\"groupby_\"+\",\".join(lis)+\"_mean\")\n        tmp = df.groupby(lis)[[target]].mean()\n        tmp.reset_index(inplace = True)\n        tmp.columns = lis + [colname]\n        df = pd.merge(df, tmp, how='left', on=lis)\n        df.loc[df[target].isnull(),target] = df.loc[df[target].isnull(),colname]\n        \n        del(df[colname])\n        lis.pop()\n    \n    return df\n    print(\"price_nan_sum : \"+str(df[\"price\"].isnull().sum()))\n    ","execution_count":8,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"863d71c48fcde117b073bce77ce49d257b5cb712","collapsed":true},"cell_type":"code","source":"df = fillna([\"category_name\",\"parent_category_name\",\"image_top_1\",\"city\"],df,\"price\")","execution_count":10,"outputs":[]},{"metadata":{"_uuid":"e05a81614ca1818628beff5e956ab7d372a92161"},"cell_type":"markdown","source":"Sorry for bad english‥\nPlease give me advice!"},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f36151e7ca592043d00ecfb1132647d88f56dd94"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}