{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#this notebook contains subset selection when the initial file is too large to load the RAM\n#subset has the same proportion of true/false value for target variable and all observations for the same ID\n#imports and params\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n#param\nfraction = 0.02\nrows_in_one_iteration = 100000\nfile_name = 'train_subsample_' + str(format(fraction, '0.0%')) + '.csv'\n\n\n#read file with labels\nraw_lebels = pd.read_csv('train_labels.csv',sep=',', header = 0)\n\n#check the data\n#check labels for empty values\nraw_lebels['target'].unique()\n\n#check if there are multiple rows for one client id\nclients_total = len(raw_lebels)\nprint('num of unique clients: ' + str(clients_total))\n\nf = raw_lebels['target'].groupby(raw_lebels['customer_ID']).transform('size').ge(2)\nprint('num of clients with more then 1 row in file: ' + str(len(raw_lebels.loc[f])))\n\n#split data to samples: train and test\n#train sample with joined target value will be loaded to file as subsample\ny_all = raw_lebels['target'].copy()\nx_train, x_test, y_train, y_test = train_test_split(raw_lebels, y_all, test_size=1-fraction, random_state=0, stratify = y_all)\n\n#check split and stratification\nprint('num of clients in train set: ' + str(len(x_train)))\nprint('num of clients in test set: ' + str(len(x_test)))\nprint('should be close to fraction variable: ' + str(len(y_train)/len(y_test)))\nprint('% of label == true in initial set: ' + str(sum(y_all)/len(y_all)))\nprint('% of label == true in target subset: ' + str(sum(y_train)/len(y_train)))\nprint('% of label == true in rest clients subset: ' + str(sum(y_test)/len(y_test)))\n\n#write to file subset of clients with only train-subset clients \ndf = pd.read_csv('train_data.csv',sep=',', header = 0, nrows=rows_in_one_iteration)\n\ndf = df.merge(x_train, left_on='customer_ID', right_on='customer_ID', how = 'right') \n\ndf.to_csv(file_name, sep=',', index=False)\n\nfor df in pd.read_csv('train_data.csv',sep=',', header = None, skiprows= rows_in_one_iteration, chunksize=rows_in_one_iteration):\n    df = df.merge(x_train, left_on=0, right_on='customer_ID', how = 'right').drop([\"customer_ID\"], axis=1)\n    df.to_csv(file_name, sep=',', index=False, mode = 'a', header=False)\n    \n#check the file saved correctly\ndf_check = pd.read_csv(file_name,sep=',', header = 0)\nf = df_check['target'].groupby(df_check['customer_ID']).size()\nprint('num of unique clients in created file: ' + str(len(f)))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]}]}