{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport os\nimport gc\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\n\nDATA_DIR = '../input/'\ntarget_col = 'deal_probability'\nos.listdir(DATA_DIR)","execution_count":23,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"usecols = ['region', 'city', 'parent_category_name', 'category_name', \n           'param_1', 'param_2', 'param_3', 'title', 'description']\ntrain = pd.read_csv(DATA_DIR+'train.csv', usecols=usecols+[target_col])\ntest = pd.read_csv(DATA_DIR+'test.csv', usecols=usecols)","execution_count":24,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"174f489dcb16262fd34a93bc5142790779078603"},"cell_type":"code","source":"train.head()","execution_count":25,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6942a5cebd475cbaca915606950b4ed4727713b5"},"cell_type":"code","source":"train['description'].isnull().sum()","execution_count":27,"outputs":[]},{"metadata":{"trusted":true,"collapsed":true,"_uuid":"f542ce46a9b003ecc757e296dd5fe79134bbcd4b"},"cell_type":"code","source":"train['description'].fillna('unknown', inplace=True)\ntest['description'].fillna('unknown', inplace=True)","execution_count":28,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b42ecd1f57d4365b0150c8c40c1acb10395a5843"},"cell_type":"code","source":"train['description'].isnull().sum()","execution_count":29,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"58bec0fdfd29c2fd263b6d5ee51237f1b4017b91","collapsed":true},"cell_type":"code","source":"train = train.fillna('')\ntest = test.fillna('')\ny = train[target_col].values\ndel train[target_col]; gc.collect()\ntrain_num = len(train)\ndf = pd.concat([train, test], ignore_index=True)\ndel train, test; gc.collect()\n\nraw_cols = df.columns.tolist()\ndf['context'] = ''\nfor c in ['parent_category_name', 'category_name', 'param_1', 'param_2', 'param_3', 'title']:\n    df[c] = df[c].str.lower()\n    df['context'] += ' ' + df[c]\ndf['context'].fillna('unknown', inplace=True)\ndf['text'] = df['description'].str.lower()\nfor c in raw_cols:\n    del df[c]\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"993ecfe456097ada8cec2cd93b69eac51c7425f5","collapsed":true},"cell_type":"code","source":"from sklearn.model_selection import KFold\nkf = KFold(n_splits=10, shuffle=True, random_state=233)\n\ndf['eval_set'] = 10 #for test\nfor fold_i, (_, test_index) in enumerate(kf.split(y)):\n    df.loc[test_index, 'eval_set'] = fold_i\ndf['label'] = 2\ndf.loc[np.arange(train_num), 'label'] = y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7fe3e8c18ef8a8433d476c9f12ab708ef97f9a3","collapsed":true},"cell_type":"code","source":"df.head(10)","execution_count":22,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5712ab1ed9e2b95e8430f03616cd4730af6587b5","collapsed":true},"cell_type":"code","source":"df.to_csv('textdata.csv', index=False)","execution_count":13,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}