{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"10bf7ff482058e6ba14cf49e1c6b43e7d8a7e091"},"cell_type":"markdown","source":"lets import few important libraries and packages"},{"metadata":{"trusted":true,"_uuid":"cae73b72b0781cf6d8f1c76f12a906b46640de95"},"cell_type":"code","source":"import seaborn as sns # for intractve graphs\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.style.use(\"ggplot\")\n%matplotlib inline\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ddc2d0401d501c0778410b4d8005c323c7564d73"},"cell_type":"markdown","source":"Reading our data using pandas"},{"metadata":{"trusted":true,"_uuid":"8e040df78f6bd6981a9adbc2728fe8dab2258952"},"cell_type":"code","source":"train_data = pd.read_csv(\"../input/train.csv\", header=0)\ntest_data = pd.read_csv(\"../input/test.csv\", header=0)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c5ec9a2da0aea9c151de8b50a9b51069bcfb565a"},"cell_type":"markdown","source":"Lets describe the data "},{"metadata":{"trusted":true,"_uuid":"73f89497f338daf2f03c9fe02202cbebef72412b"},"cell_type":"code","source":"train_data.info()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"588420d3b4c3e20ff9375afcffa1da2dd9a0f17d"},"cell_type":"markdown","source":"We have 1306122 rows of data and lets see if we have any missing values"},{"metadata":{"trusted":true,"_uuid":"363c3a68f38e057fdb6807ba2abace3f8f4d02c6"},"cell_type":"code","source":"train_data.isnull().sum()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4097862a1fe2dd998115e99b66e63f608516f934"},"cell_type":"markdown","source":"Good!! we don't have any missing values, don't have to handle them now. Moving further lets see if have class unbalance problem"},{"metadata":{"trusted":true,"_uuid":"30cd3f223a8283130989d5a5432c16bd24e379d7"},"cell_type":"code","source":"sns.countplot(\"target\",data=train_data)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"472e52c1c6605ce415cd4ad9747153fd2faf7b0b"},"cell_type":"markdown","source":"Ohh my gosh!! we have class imbalance problem, need to balance it otherwise we will get skewed results."},{"metadata":{"trusted":true,"_uuid":"7959630719e60d7b9432d08c54c1ebb61f84c219"},"cell_type":"code","source":"# now let us check in the number of Percentage\nCount_Normal_transacation = len(train_data[train_data[\"target\"]==0]) # normal transaction are repersented by 0\nCount_insincere_transacation = len(train_data[train_data[\"target\"]==1]) # fraud by 1\nPercentage_of_Normal_transacation = Count_Normal_transacation/(Count_Normal_transacation+Count_insincere_transacation)\nprint(\"percentage of normal transacation is\",Percentage_of_Normal_transacation*100)\nPercentage_of_insincere_transacation= Count_insincere_transacation/(Count_Normal_transacation+Count_insincere_transacation)\nprint(\"percentage of fraud transacation\",Percentage_of_insincere_transacation*100)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cffc68366ecde41c210efd320815d5c2f6fd4616"},"cell_type":"markdown","source":"Now handling the class imbalance problem, we are using a function to tackle this, so just to save some processing time we would be taking an assumption that the training data is class=1 is equal to class=0, means we take equal number of data of both clases."},{"metadata":{"trusted":true,"_uuid":"f163e93bfc420fd85ac2fe25de61b61d409c6a19"},"cell_type":"code","source":"insincere_indices= np.array(train_data[train_data.target==1].index)\nnormal_indices = np.array(train_data[train_data.target==0].index)\n#now let us a define a function for make undersample data with different proportion\n#different proportion means with different proportion of normal classes of data\ndef undersample(normal_indices,insincere_indices,times):#times denote the normal data = times*fraud data\n    Normal_indices_undersample = np.array(np.random.choice(normal_indices,(times*Count_insincere_transacation),replace=False))\n    print(len(Normal_indices_undersample))\n    undersample_data= np.concatenate([insincere_indices,Normal_indices_undersample])\n\n    undersample_data = train_data.iloc[undersample_data,:]\n    #print(undersample_data)\n    print(len(undersample_data))\n\n    print(\"the normal transacation proportion is :\",len(undersample_data[undersample_data.target==0])/len(undersample_data))\n    print(\"the fraud transacation proportion is :\",len(undersample_data[undersample_data.target==1])/len(undersample_data))\n    print(\"total number of record in resampled data is:\",len(undersample_data))\n    return(undersample_data)\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f050760cb66eeb8daadb376e54d1bb9bb83d5840"},"cell_type":"markdown","source":"You can handle change the last parameter of the function for changing the proportion of the classes, as of now I have taken equal proportion."},{"metadata":{"trusted":true,"_uuid":"e2d0c05fb8dbf435d8a609dad13eb8ef0a5e4d13"},"cell_type":"code","source":"Undersample_data = undersample(normal_indices,insincere_indices,1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cbd6806a476a7dceda61441dff7c19aeaad8865f"},"cell_type":"code","source":"questions=Undersample_data.iloc[:,1].values.tolist()\nlabels=Undersample_data.iloc[:,2].values.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b6ce2c1591867da654f46cfc511a3ae11b4eb9a8"},"cell_type":"markdown","source":"**Now we can proceed with handing NLP and Embeddings.**"},{"metadata":{"trusted":true,"_uuid":"62173225ad0dc6a1dc9fd3c8184c38f82f7c1b3c"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"938d1ff8b5d7d3bedbb0b67ac9d23595fa4cfc55"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}