{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"# def plot_line_mat(k, xoff=0, yoff=0):\n#     fit, axs = plt.subplots(nrows=2, ncols=2, figsize=(15, 8.6), dpi=100)\n#     _mean=gps[k].mean().drop(keys - {k}, axis=1)\n#     _var=gps[k].var().drop(keys-{k}, axis=1)\n\n#     for i in range(4):\n#         if i != 1:    \n#             axs[0, 0].plot(_mean.index, _mean[f'input_{i}'], marker='.');\n#             axs[1, 0].plot(_var.index, _var[f'input_{i}'], marker='.');\n#     axs[0, 0].set_title(f'mean - by {k}')\n#     axs[1, 0].set_title(f'var - by {k}')\n\n#     sns.heatmap(_mean.corr(), annot=True, ax=axs[0, 1]);\n#     axs[0, 1].set_title('correlation - mean')\n#     sns.heatmap(_var.corr(), annot=True, ax=axs[1, 1]);\n#     axs[1, 1].set_title('correlation - var');\n#     plt.savefig(f'line_correlation_{k}')\n\n    \ndef plot_line_mat_all(lis,year=2018):\n    for k in lis:\n        fig, axs = plt.subplots(nrows=2, ncols=2, figsize=(15, 8.6), dpi=150)\n        _mean=gps[k].mean().drop(keys - {k}, axis=1)\n        _var=gps[k].var().drop(keys-{k}, axis=1)\n\n        for i in range(4):\n            if i != 1:    \n                axs[0, 0].plot(_mean.index, _mean[f'input_{i}'], marker='.');\n                axs[1, 0].plot(_var.index, _var[f'input_{i}'], marker='.');\n        axs[0, 0].set_title(f'mean - by {k}')\n        axs[0, 0].legend(_mean.columns, framealpha=.3, loc='best', bbox_to_anchor=(0.5, 0., 0.5, 0.5))\n        axs[1, 0].set_title(f'var - by {k}')\n        axs[1, 0].legend(_var.columns, framealpha=.3, loc='best', bbox_to_anchor=(0.5, 0., 0.5, 0.5))\n        \n        sns.heatmap(_mean.corr(), annot=True, ax=axs[0, 1]);\n        axs[0, 1].set_title('corr - mean')\n        sns.heatmap(_var.corr(), annot=True, ax=axs[1, 1]);\n        axs[1, 1].set_title('corr - var');\n        fig.suptitle(f'({year}) by {k} - mean & var')\n        fig.savefig(f'line_correlation_{k}')\n    plt.show()\n\n\ndef reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df\n\ndef is_weekend(num):\n    return num > 5\n\ndef date_expand(df, pipeline = True):\n    df['date'] = pd.to_datetime( df.date )\n    dt = df['date'].dt\n    df['month'] = dt.month\n    df['day'] = dt.day\n    df['year'] =  dt.year\n    df['week'] = dt.week\n    df['weekday'] = dt.weekday\n    df['weekofyear'] = dt.weekofyear\n    df['weekend'] = dt.weekday.apply(is_weekend)\n#     df['pastdays'] = df.input\n    return df if pipeline else None","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport gc\nimport re\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport glob\nimport itertools\nfrom collections import Counter\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    continue\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 5GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nfrom sklearn.impute import SimpleImputer","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"year_file = {}\nl=glob.glob('../input/kddbr-2020/*.csv')\nfor i in l:\n    year = re.search('\\d{4}', i.split('/')[-1])\n    if not year:\n        print(i)\n        continue\n    year = int(year.group())\n    year_file[year] = year_file.get(year, []) + [i]\ndel year\nyear_file.keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for k in [f'input_{i}' for i in range(4)]:\n    print(df18[k].value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ndf18 = []\nfor i in year_file[2018]:\n    df = pd.read_csv(i)\n    df18.append(df)\n    del df\ndf18 = pd.concat(df18)\ndate_expand(df18, False);","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Colunas de interesse para mostrar correlação","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\ncs = ['input_0', 'input_2', 'input_3']\nkeys = {'day', 'month', 'weekday', 'weekofyear'}\n\ndzip = df18[cs + list(keys)]\n\ngps = {}\n\nfor k in keys:\n    gps[k] = dzip.groupby(k)\nlk = list(keys)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Counter(df18.drop(['id', 'date'] + [f'input_{i}' for i in range(4)], axis=1).columns)\n# df18.drop(['id', 'date'] + [f'input_{i}' for i in range(4)], axis=1).groupby()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# fit, axs = plt.subplots(nrows=2, ncols=2, figsize=(15, 8.6), dpi=100)\n# k ='day'\n# _mean=gps['day'].mean().drop(['month', 'weekday', 'weekofyear'], axis=1)\n# _var=gps['day'].var().drop(['month', 'weekday', 'weekofyear'], axis=1)\n\n# for i in range(4):\n#     if i != 1:    \n#         axs[0, 0].plot(_mean.index, _mean[f'input_{i}'], marker='.');\n#         axs[1, 0].plot(_var.index, _var[f'input_{i}'], marker='.');\n# axs[0, 0].set_title(f'mean - by {k}')\n# axs[1, 0].set_title(f'var - by {k}')\n\n# sns.heatmap(_mean.corr(), annot=True, ax=axs[0, 1]);\n# axs[0, 1].set_title('correlation - mean')\n# sns.heatmap(_var.corr(), annot=True, ax=axs[1, 1]);\n# axs[1, 1].set_title('correlation - var');\n# plt.savefig(f'line_correlation_{k}')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_line_mat_all(lk)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\npast_in = {}\npast_out = {}\nfor i in range(1, 15):\n    past_in[i] = df18.filter(regex=fr'^in.+?\\d+_{i}$', axis=1)\n\nfor i in range(0, 7):\n    past_out[i] = df18.filter(regex=fr'^out.+?\\d+_{i}$', axis=1)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipt = SimpleImputer(strategy='mean')\n\ndl = pd.DataFrame(ipt.fit_transform(past_in[1].T).T)\ndl","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pi.any()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pi = past_in[1]\npi.loc[:, pi.isnull().sum() > 0]\n# pi.shape, pi.isnull().sum().shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"past_out[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipt = SimpleImputer(strategy='mean')\n\npd.DataFrame(ipt.fit_transform(past_out[1]), columns=past_out[1].columns)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}