{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!git clone https://github.com/valteresj2/features-analysis-individual.git\n!pip install xlsxwriter","metadata":{"execution":{"iopub.status.busy":"2022-07-26T19:01:31.547769Z","iopub.execute_input":"2022-07-26T19:01:31.548517Z","iopub.status.idle":"2022-07-26T19:01:47.402224Z","shell.execute_reply.started":"2022-07-26T19:01:31.548466Z","shell.execute_reply":"2022-07-26T19:01:47.401137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## For it is experiment i will apply the bivariate analysis, i create a function to generated it is analyse, follow my github (https://github.com/valteresj2/features-analysis-individual.git)\n## this is dataset to the experiment i did create in modified the transformation on my publish (https://www.kaggle.com/code/valteresj/create-1k-features-with-datatable) where created approximately 1.5k features.","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom datatable import (dt, f, by, ifelse, update, sort,join,\n                       count, min, max, mean, sum, rowsum,sd,last,median)\nimport os\nimport glob\nfrom random import sample\nos.chdir('/kaggle/working/features-analysis-individual')\nfrom fai_class import FAI","metadata":{"execution":{"iopub.status.busy":"2022-07-26T19:01:49.626595Z","iopub.execute_input":"2022-07-26T19:01:49.627036Z","iopub.status.idle":"2022-07-26T19:01:49.634809Z","shell.execute_reply.started":"2022-07-26T19:01:49.626980Z","shell.execute_reply":"2022-07-26T19:01:49.633894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## To download of dataset follow the link (https://www.kaggle.com/datasets/valteresj/train-amex-1k-features)","metadata":{}},{"cell_type":"code","source":"df_train = dt.fread('/kaggle/input/train-amex-1k-features/train_feats2.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:05:10.937702Z","iopub.execute_input":"2022-07-26T18:05:10.938000Z","iopub.status.idle":"2022-07-26T18:06:11.868347Z","shell.execute_reply.started":"2022-07-26T18:05:10.937973Z","shell.execute_reply":"2022-07-26T18:06:11.865747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## I random select 50% of full data, the motivate he was it is mermory ram kernel don't support all dataset. ","metadata":{}},{"cell_type":"code","source":"ind=sample(list(range(0,df_train.shape[0])),int(df_train.shape[0]*0.5))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:11.871186Z","iopub.execute_input":"2022-07-26T18:06:11.871612Z","iopub.status.idle":"2022-07-26T18:06:12.176721Z","shell.execute_reply.started":"2022-07-26T18:06:11.871576Z","shell.execute_reply":"2022-07-26T18:06:12.175304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The next step is convert Frame to DataFrame","metadata":{}},{"cell_type":"code","source":"df_train=df_train[ind,:].to_pandas()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:12.178730Z","iopub.execute_input":"2022-07-26T18:06:12.179233Z","iopub.status.idle":"2022-07-26T18:06:29.308438Z","shell.execute_reply.started":"2022-07-26T18:06:12.179188Z","shell.execute_reply":"2022-07-26T18:06:29.307266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Then is convert the target numeric to string","metadata":{}},{"cell_type":"code","source":"df_train['target']=df_train['target'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:29.312493Z","iopub.execute_input":"2022-07-26T18:06:29.312955Z","iopub.status.idle":"2022-07-26T18:06:29.433488Z","shell.execute_reply.started":"2022-07-26T18:06:29.312910Z","shell.execute_reply":"2022-07-26T18:06:29.432476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Below is converted the categorical features the numeric to string.","metadata":{}},{"cell_type":"code","source":"cat_features = ['B_30','B_38','D_114','D_116','D_117','D_120','D_126','D_63','D_64','D_66','D_68']\nfor i in cat_features:\n    var=[j for j in df_train.columns if j.find(i)>=0][0]\n    df_train[var]=df_train[var].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:29.434806Z","iopub.execute_input":"2022-07-26T18:06:29.435257Z","iopub.status.idle":"2022-07-26T18:06:36.261376Z","shell.execute_reply.started":"2022-07-26T18:06:29.435217Z","shell.execute_reply":"2022-07-26T18:06:36.260224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## A correction small is convert bool to string.","metadata":{}},{"cell_type":"code","source":"g=df_train.dtypes\nfor i in g.index:\n    if g[i]==bool:\n        df_train[i]=df_train[i].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:36.263105Z","iopub.execute_input":"2022-07-26T18:06:36.263542Z","iopub.status.idle":"2022-07-26T18:06:36.571728Z","shell.execute_reply.started":"2022-07-26T18:06:36.263496Z","shell.execute_reply":"2022-07-26T18:06:36.570595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Convert inf and -inf in Nan.","metadata":{}},{"cell_type":"code","source":"df_train.replace([np.inf, -np.inf], np.nan, inplace=True)\ndf_train['B_31_SHIFTMAX']=df_train['B_31_SHIFTMAX'].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-07-26T18:06:36.573225Z","iopub.execute_input":"2022-07-26T18:06:36.573584Z","iopub.status.idle":"2022-07-26T18:06:40.165964Z","shell.execute_reply.started":"2022-07-26T18:06:36.573552Z","shell.execute_reply":"2022-07-26T18:06:40.164871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Below is applied the function Features Analysis Individual","metadata":{}},{"cell_type":"code","source":"performance=FAI(dt=df_train.drop(columns=['customer_ID','DATE_MAX']),target='target',path='/kaggle/working')\nget_result_biv=performance.perf_features()","metadata":{"execution":{"iopub.status.busy":"2022-07-26T19:02:01.929605Z","iopub.execute_input":"2022-07-26T19:02:01.929992Z","iopub.status.idle":"2022-07-26T19:51:55.687983Z","shell.execute_reply.started":"2022-07-26T19:02:01.929959Z","shell.execute_reply":"2022-07-26T19:51:55.687092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## The result is generate a excel file with all features with show the table below.","metadata":{}},{"cell_type":"code","source":"get_result_biv['P_2_LAST']","metadata":{"execution":{"iopub.status.busy":"2022-07-26T19:52:36.183227Z","iopub.execute_input":"2022-07-26T19:52:36.183702Z","iopub.status.idle":"2022-07-26T19:52:36.206116Z","shell.execute_reply.started":"2022-07-26T19:52:36.183662Z","shell.execute_reply":"2022-07-26T19:52:36.205297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Is calculated the metric KS - Kolmogorov Smirnov, Ewntropy and Relative Risk, coming soon i will go add customize performance function by user.\n## By example the feature 'P_2_LAST' is that show the maximun value of KS in between all features created.\n","metadata":{}},{"cell_type":"markdown","source":"## With it is dataset my result then 79.1, coming soon i will go post the table of test. \n## Any questions put it in the comments","metadata":{}}]}