{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import gc\nimport os\nimport numpy as np\nimport pandas as pd\nimport subprocess\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport xgboost as xgb","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Introduction\nThere are some interesting facts that this competition attracts me.\n- Data size is very large:240 million rows\n- Imbalanced data\n- Use ROC-AUC as performance metrics\n- Real life problem\n"},{"metadata":{},"cell_type":"markdown","source":"# Load Data"},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_fsize(dpath,s=30):\n    \"\"\"check file size\n    Args:\n    dpath: file directory\n    s: string length in total after padding\n    \n    Returns:\n    None\n    \"\"\"\n    for f in os.listdir(dpath):\n        print(f.ljust(s) + str(round(os.path.getsize(dpath+'/' + f) / 1000000, 2)) + 'MB')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"check_fsize('../input')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_fline(fpath):\n    \"\"\"check total number of lines of file for large files\n    \n    Args:\n    fpath: string. file path\n    \n    Returns:\n    None\n    \n    \"\"\"\n    lines = subprocess.run(['wc', '-l', fpath], stdout=subprocess.PIPE).stdout.decode('utf-8')\n    print(lines, end='', flush=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fs=['../input/train.csv', '../input/test.csv', '../input/train_sample.csv']\n[check_fline(s) for s in fs]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Total line number is huge so just load 1 million row of data for analysis"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Load sample training data\ndf_train = pd.read_csv('../input/train.csv', nrows=1000000, parse_dates=['click_time'])\ndf_test = pd.read_csv('../input/test.csv', nrows=1000000, parse_dates=['click_time'])\n\n# Show head\nprint(df_train.head())\n\n# show shape\nprint(df_test.head())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check feature unique value"},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_cunique(df,cols):\n    \"\"\"check unique values for each column\n    df: data frame. \n    cols: list. The columns of data frame to be counted\n    \"\"\"\n    df_nunique = df[cols].nunique().to_frame()\n    df_nunique = df_nunique.reset_index().rename(columns={'index': 'feat',0:'nunique'})\n    return df_nunique","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_nunique = check_cunique(df_train,['ip', 'app', 'device', 'os', 'channel'])\ndf_nunique","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(15, 8))\nsns.set(font_scale=1.2)\nsns.barplot(x=\"feat\" ,y=\"nunique\", data=df_nunique,log=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def feat_value_count(df,colname):\n    \"\"\"value count of each feature\n    \n    Args\n    df: data frame.\n    colname: string. Name of to be valued column\n    \n    Returns\n    df_count: data frame.\n    \"\"\"\n    df_count = df[colname].value_counts().to_frame().reset_index()\n    df_count = df_count.rename(columns={'index':colname+'_values',colname:'counts'})\n    return df_count","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"feat_value_count(df_train,'is_attributed')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Check missing value"},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_missing(df,cols=None,axis=0):\n    \"\"\"check data frame column missing situation\n    Args\n    df: data frame.\n    cols: list. List of column names\n    \n    Returns\n    missing_info: data frame. \n    \"\"\"\n    if cols != None:\n        df = df[cols]\n    missing_num = df.isnull().sum(axis).to_frame().rename(columns={0:'missing_num'})\n    missing_num['minssing_percent'] = df.isnull().mean(axis)*100\n    return missing_num.sort_values(by='minssing_percent',ascending = False) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(check_missing(df_train))\nprint(check_missing(df_train,axis=1).head())","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}