{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><h1>Manifold Representation: UMAP</h1></center>\n<center><h4>More info <a href='https://github.com/lmcinnes/umap'>here</a>.</h4></center>","metadata":{}},{"cell_type":"code","source":"%reset -sf\n\nfrom IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = 'all'","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-16T05:22:03.581859Z","iopub.execute_input":"2022-07-16T05:22:03.582584Z","iopub.status.idle":"2022-07-16T05:22:03.713747Z","shell.execute_reply.started":"2022-07-16T05:22:03.582477Z","shell.execute_reply":"2022-07-16T05:22:03.712500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip show umap-learn","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-16T05:22:03.716599Z","iopub.execute_input":"2022-07-16T05:22:03.717300Z","iopub.status.idle":"2022-07-16T05:22:15.829719Z","shell.execute_reply.started":"2022-07-16T05:22:03.717251Z","shell.execute_reply":"2022-07-16T05:22:15.828201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data\n\nfrom pandas import read_csv\n\ntrain = read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv')\ntrain","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:15.832076Z","iopub.execute_input":"2022-07-16T05:22:15.832571Z","iopub.status.idle":"2022-07-16T05:22:17.251655Z","shell.execute_reply.started":"2022-07-16T05:22:15.832526Z","shell.execute_reply":"2022-07-16T05:22:17.250531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Mem\n\ndef reduce_memory(data, verbose=True):\n    import numpy as np\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = data.memory_usage().sum() / 1024 ** 2\n    for column in data.columns:\n        column_type = data[column].dtypes\n        if column_type in numerics:\n            c_min = data[column].min()\n            c_max = data[column].max()\n            if str(column_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    data[column] = data[column].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    data[column] = data[column].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    data[column] = data[column].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    data[column] = data[column].astype(np.int64)\n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    data[column] = data[column].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    data[column] = data[column].astype(np.float32)\n                else:\n                    data[column] = data[column].astype(np.float64)\n    end_mem = data.memory_usage().sum() / 1024 ** 2\n    if verbose:\n        print('Memory usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100*(start_mem - end_mem) / start_mem))\n    return data\n\ntrain = reduce_memory(train)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:17.254943Z","iopub.execute_input":"2022-07-16T05:22:17.256123Z","iopub.status.idle":"2022-07-16T05:22:17.372092Z","shell.execute_reply.started":"2022-07-16T05:22:17.256082Z","shell.execute_reply":"2022-07-16T05:22:17.370708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Normalize data\n\nfrom sklearn.preprocessing import StandardScaler\nfrom pandas import Series, DataFrame\n\nstd = StandardScaler()\ntrain = DataFrame(std.fit_transform(train), columns=train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:17.374747Z","iopub.execute_input":"2022-07-16T05:22:17.375153Z","iopub.status.idle":"2022-07-16T05:22:17.885079Z","shell.execute_reply.started":"2022-07-16T05:22:17.375103Z","shell.execute_reply":"2022-07-16T05:22:17.883832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Straight to business\n\nfrom umap import UMAP\n\nmanif = UMAP(n_components=2)\ntrain_ = manif.fit_transform(train.sample(frac=0.1))\ntrain_","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:22:17.886631Z","iopub.execute_input":"2022-07-16T05:22:17.887679Z","iopub.status.idle":"2022-07-16T05:23:23.573194Z","shell.execute_reply.started":"2022-07-16T05:22:17.887622Z","shell.execute_reply":"2022-07-16T05:23:23.571727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vis\n\nimport matplotlib.pyplot as plt\n\nfig, ax = plt.subplots(figsize=(10,7))\n_ = ax.scatter(train_[:, 0], train_[:, 1], alpha=0.25)\n\n# As you can see, there doesn't seem to be groupings...\n\n# Waldemar suggested that groupings are overlapping, so they are not clearly visible\n\n# However, I don't think that is correct as the whole point of Umap group\n# similarities and separate differences, visibly!\n\n# Anyway, let's do a 4-var umap in a colored 3d graph ","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:23.574931Z","iopub.execute_input":"2022-07-16T05:23:23.576017Z","iopub.status.idle":"2022-07-16T05:23:23.894557Z","shell.execute_reply.started":"2022-07-16T05:23:23.575973Z","shell.execute_reply":"2022-07-16T05:23:23.893617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Straight to business\n\nfrom umap import UMAP\n\nmanif = UMAP(n_components=4)\ntrain_ = manif.fit_transform(train.sample(frac=0.1))\ntrain_","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:23.895995Z","iopub.execute_input":"2022-07-16T05:23:23.896550Z","iopub.status.idle":"2022-07-16T05:23:44.728447Z","shell.execute_reply.started":"2022-07-16T05:23:23.896512Z","shell.execute_reply":"2022-07-16T05:23:44.727311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 3D graph\n\nfrom plotly_express import scatter_3d\n\nfig = scatter_3d(train_, \n                 x=train_[:, 0], \n                 y=train_[:, 1], \n                 z=train_[:, 2], \n                 size=[5]*len(train_),\n                 color=train_[:, 3]\n                )\nfig.update_layout(margin_b=0, margin_t=0, margin_l=0, margin_r=0, )\nfig.show()\n\n# Still not quite apparent groupings...","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:44.729949Z","iopub.execute_input":"2022-07-16T05:23:44.730276Z","iopub.status.idle":"2022-07-16T05:23:47.965922Z","shell.execute_reply.started":"2022-07-16T05:23:44.730247Z","shell.execute_reply":"2022-07-16T05:23:47.964762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The above graph made me think an idea\n\nfrom umap import UMAP\nfrom string import ascii_lowercase\n\nmanif = UMAP(n_components=14)\ntrain_ = manif.fit_transform(train.drop('id', axis=1).sample(frac=0.05))\ntrain_ = DataFrame(train_, columns=list(ascii_lowercase[:train_.shape[1]]))\ntrain_","metadata":{"execution":{"iopub.status.busy":"2022-07-16T05:23:47.968958Z","iopub.execute_input":"2022-07-16T05:23:47.970204Z","iopub.status.idle":"2022-07-16T05:24:02.719123Z","shell.execute_reply.started":"2022-07-16T05:23:47.970166Z","shell.execute_reply":"2022-07-16T05:24:02.717855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}