{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Summary\nHere's everything we need to know about the data distribution in one \"simplified\" set of plots. However, with great power come great confusion, so I'll clarify a bit.\n\n+ Each chart shows information on a single feature, for the class as a whole *and* separated by product code.\n+ For continuous features, he bins are *not* equal width, but instead are partitioned by quantiles so that there are a roughly equal number of samples (300 or so) per bin.\n+ The pastel histograms show the number of bin instances *per product*. \n+ The leftmost and rightmost bins have been removed from the graphs, because they end up being absurdly wide. This makes them both misleading and really ugly when included on the graphs.\n+ With 5 products, we'll have an average of 60 or so samples per bin but, as you can see, the different products have skewed distributions, so there may be as few as 20 in a product bin. (Counts are shown on the left.)\n+ We show scatter plots of the failure rate for each bin by product (colored dots) and for the dataset as a whole (black).\n+ We show a line graph reflecting the same data. However, it is *smoothed* by using larger bins. They're 8 times larger, so we have between 160 and 1600 samples per smoothed bin.\n\nThere are some interesting sights here:\n+ It becomes clear just how much consistently higher the failure rate is for product A and lower for product B.\n+ We can see that the trend line for measurement_17 is a lot more muddled than one might think.\n+ We can see how skewed some of the features are across different products.\n+ Etc.","metadata":{}},{"cell_type":"code","source":"import warnings\nfrom contextlib import contextmanager\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport matplotlib.axes as axes\nfrom typing import Generator\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-06T22:58:19.874815Z","iopub.execute_input":"2022-08-06T22:58:19.875263Z","iopub.status.idle":"2022-08-06T22:58:19.881965Z","shell.execute_reply.started":"2022-08-06T22:58:19.875228Z","shell.execute_reply":"2022-08-06T22:58:19.880786Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Xy_train = pd.read_csv(\"../input/tabular-playground-series-aug-2022/train.csv\", index_col='id')\ny_train = Xy_train.failure\nX_train = Xy_train.drop(columns=[\"failure\"])\nX_test = pd.read_csv(\"../input/tabular-playground-series-aug-2022/test.csv\", index_col='id')\nX_comb = pd.concat([X_train, X_test], axis=\"rows\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-06T22:53:44.170612Z","iopub.execute_input":"2022-08-06T22:53:44.171329Z","iopub.status.idle":"2022-08-06T22:53:44.562630Z","shell.execute_reply.started":"2022-08-06T22:53:44.171288Z","shell.execute_reply":"2022-08-06T22:53:44.561592Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mpl.rcParams[\"patch.linewidth\"] = 0.2\nmpl.rcParams[\"patch.force_edgecolor\"] = True\nmpl.rcParams[\"patch.edgecolor\"] = \"w\"\ncycle_primary = (plt.cycler(alpha=[1, .6, .4]) * plt.cycler(linestyle=['-', ':', '--', '-.']) *\n                 plt.cycler(color=\"bgrcmyk\"))\n\n@contextmanager\ndef axer(figsize=(12, 8), subplot=(1, 1, 1), many_lines=False, show_legend=False,\n         tight_layout=False, **xargs) -> Generator[axes.Axes, None, None]:\n    fig = plt.figure(0, figsize, dpi=144)\n    ax: axes.Axes = fig.add_subplot(*subplot)\n    if many_lines:\n        ax.set(prop_cycle=cycle_primary)\n    ax.set(**xargs)\n    yield ax\n    if show_legend:\n        ax.legend()\n    if tight_layout:\n        fig.tight_layout()\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-08-06T22:53:44.564577Z","iopub.execute_input":"2022-08-06T22:53:44.565413Z","iopub.status.idle":"2022-08-06T22:53:44.577466Z","shell.execute_reply.started":"2022-08-06T22:53:44.565365Z","shell.execute_reply":"2022-08-06T22:53:44.576329Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = 2\ncols = [len(X_test.columns) // num_cols + 1, num_cols]\nbin_max = 6\ncolors = [\"b\", \"chartreuse\", \"brown\", \"magenta\", \"c\", \"k\"]\nbincount = 64  # 16\n\nfor i, col in enumerate(X_test.columns):\n    with axer(subplot=[*cols, i + 1], tight_layout=True, figsize=(12, 72), title=col) as ax:\n        ax2 = ax.twinx()\n        ax2.set_ylim(0, 0.4)\n        ax2.grid(True)\n        ax2.tick_params(axis='y', labelcolor='m', grid_linewidth=0.7)\n        with warnings.catch_warnings():  # ignore divide by zero for empty bins\n            warnings.filterwarnings('ignore', category=RuntimeWarning)\n\n            if Xy_train[col].nunique() >= bin_max:\n                bins = np.quantile(Xy_train[col].dropna().values, np.linspace(0, 1, bincount))\n                # bins = np.linspace(Xy_train[col].min() - 0.01, Xy_train[col].max(), 15)\n                smbins = np.quantile(Xy_train[col].dropna().values,\n                                     np.linspace(0, 1, bincount // 8))\n\n            groups = Xy_train.groupby(\"product_code\")\n\n            def plot_df(idx, groupname, df):\n                style = \"solid\" if groupname != \"all\" else \"dashed\"\n                data = df[col]\n                uniq = data.nunique()\n                if uniq < bin_max:\n                    counts = data.value_counts()\n                    if groupname != \"all\":\n                        ax.bar(counts.index, counts, width=1, alpha=0.1, color=colors[idx])\n\n                    toplot = df.assign(data=data).groupby(data).failure\n                    mean = toplot.mean()\n                    if (len(mean) < 2) or (groupname == \"all\"):\n                        ax2.scatter(mean.index, mean, s=20, color=colors[idx], label=groupname)\n                    else:\n                        ax2.plot(mean.index, mean, linewidth=2, alpha=.5, color=colors[idx],\n                                 label=groupname)\n                else:\n                    ax.set_xlim(bins[1], bins[-2])\n                    if groupname != \"all\":\n                        ax.hist(data, bins=bins, alpha=0.1, color=colors[idx])\n\n                    total, _ = np.histogram(data, bins=bins)\n                    failures, _ = np.histogram(data[df.failure == 1], bins=bins)\n                    xaxis = np.mean([bins[1:], bins[:-1]], axis=0)\n                    yaxis = (failures+0.2) / (total+1)\n                    totalsm, _ = np.histogram(data, bins=smbins)\n                    failuressm, _ = np.histogram(data[df.failure == 1], bins=smbins)\n                    xaxissm = np.mean([smbins[1:], smbins[:-1]], axis=0)\n                    yaxissm = (failuressm+0.2) / (totalsm+1)\n\n                    ax2.plot(xaxissm, yaxissm, linewidth=1, alpha=1, color=colors[idx],\n                             label=groupname, linestyle=style)\n                    ax2.scatter(xaxis, yaxis, alpha=1, s=10, color=colors[idx])\n\n            for g, (product, df) in enumerate(groups):\n                plot_df(g, product, df)\n            plot_df(5, \"all\", Xy_train)\n        ax2.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T22:53:44.579922Z","iopub.execute_input":"2022-08-06T22:53:44.580764Z","iopub.status.idle":"2022-08-06T22:54:22.673322Z","shell.execute_reply.started":"2022-08-06T22:53:44.580714Z","shell.execute_reply":"2022-08-06T22:54:22.670891Z"},"jupyter":{"source_hidden":true},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}