{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <span style=\"color:DarkCyan\">Incremental Principal Component Analysis (IPCA)</span> | American Express - Default Prediction","metadata":{"execution":{"iopub.status.busy":"2022-06-14T17:15:32.069337Z","iopub.execute_input":"2022-06-14T17:15:32.069987Z","iopub.status.idle":"2022-06-14T17:15:32.075097Z","shell.execute_reply.started":"2022-06-14T17:15:32.069954Z","shell.execute_reply":"2022-06-14T17:15:32.074480Z"}}},{"cell_type":"markdown","source":"Thank you for viewing my notebook, I hope you enjoy it 📊<br>\nDon't hesitate to leave any feedback 😉","metadata":{}},{"cell_type":"markdown","source":"# Overview","metadata":{}},{"cell_type":"markdown","source":"<span style=\"font-size:22px\"><span style=\"color:DarkCyan\">Incremental Principal Component Analysis (IPCA)</span> is an alternative for principal component analysis (PCA)<br>\n<span style=\"color:DarkCyan\">IPCA</span> allows us to decompose large datasets that cannot be fit in typical PCA.<br><br>\nThe size of the original dataset is 16GB. We can read the dataset with chunks and fit in IPCA<br><br></span>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pickle as pk\nfrom sklearn.decomposition import IncrementalPCA\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"06d54df7-f433-4766-a89e-4045a7ae87cf","_cell_guid":"310fd0d8-fcee-491e-897c-18dd25b8f2f4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:03:40.719833Z","iopub.execute_input":"2022-06-14T17:03:40.720803Z","iopub.status.idle":"2022-06-14T17:03:41.409340Z","shell.execute_reply.started":"2022-06-14T17:03:40.720701Z","shell.execute_reply":"2022-06-14T17:03:41.408518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define dtypes for the pandas DataFrame","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/amex-default-prediction'\ntrain_path = os.path.join(path, 'train_data.csv')\ntrain_df = pd.read_csv(train_path, nrows=100_000)\n\nbools = train_df.select_dtypes(include=[int])\nfloats = train_df.select_dtypes(include=[float])\ncategorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\npca_cols = set(train_df.columns) - set(categorical_cols) - set(['customer_ID', 'S_2'])\ndel train_df\n\ndtypes = dict(zip(floats, [np.float32]*len(floats)))\ndtypes.update(dict(zip(bools, [bool]*len(bools))))\ndtypes.update(dict(zip(categorical_cols, ['category']*len(categorical_cols))))","metadata":{"_uuid":"a317a462-e4f1-4700-b32a-130d865bcf03","_cell_guid":"c14103cf-9e56-4642-823f-18477f104bfc","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:03:41.410803Z","iopub.execute_input":"2022-06-14T17:03:41.411112Z","iopub.status.idle":"2022-06-14T17:03:48.429729Z","shell.execute_reply.started":"2022-06-14T17:03:41.411084Z","shell.execute_reply":"2022-06-14T17:03:48.428838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create pandas DataFrame iterator - read data in chunks","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(train_path, chunksize=10**2, usecols=pca_cols, dtype=dtypes, iterator=True)","metadata":{"_uuid":"5d107647-93c3-4e2f-add6-6484574dc9af","_cell_guid":"8ad5477f-6b21-4064-ba85-6a6be5438b2f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:05.997474Z","iopub.execute_input":"2022-06-14T17:04:05.998504Z","iopub.status.idle":"2022-06-14T17:04:06.009803Z","shell.execute_reply.started":"2022-06-14T17:04:05.998393Z","shell.execute_reply":"2022-06-14T17:04:06.009046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Initialize Incremental PCA","metadata":{}},{"cell_type":"code","source":"ipca = IncrementalPCA(batch_size=10)","metadata":{"_uuid":"75d4d691-f457-44fa-930b-148c2df51fdf","_cell_guid":"053a36df-e114-4ada-94a5-e4ad4a694c81","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:10.721086Z","iopub.execute_input":"2022-06-14T17:04:10.721815Z","iopub.status.idle":"2022-06-14T17:04:10.727832Z","shell.execute_reply.started":"2022-06-14T17:04:10.721764Z","shell.execute_reply":"2022-06-14T17:04:10.726941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fit Incremental PCA with chunks","metadata":{"_uuid":"d16cc523-4453-45ea-83fc-044da024c150","_cell_guid":"8f19b646-e64a-40ab-8ed9-439d4df3dbec","jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"for chunk in df:\n    ipca.partial_fit(chunk.fillna(0))","metadata":{"_uuid":"a9b860f6-af50-46df-bd84-b75d1633df3d","_cell_guid":"86fea696-c36f-48ae-95dc-669fc11d3868","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:11.106247Z","iopub.execute_input":"2022-06-14T17:04:11.107253Z","iopub.status.idle":"2022-06-14T17:04:58.042574Z","shell.execute_reply.started":"2022-06-14T17:04:11.107211Z","shell.execute_reply":"2022-06-14T17:04:58.041472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Visualize results: Explained Variance Ratio","metadata":{"_uuid":"583ca829-4ffa-4eda-a348-79beff4d82a5","_cell_guid":"b77c4718-636e-4c88-8b16-bbccef5badd6","execution":{"iopub.status.busy":"2022-06-14T16:54:08.770713Z","iopub.execute_input":"2022-06-14T16:54:08.771260Z","iopub.status.idle":"2022-06-14T16:54:08.776781Z","shell.execute_reply.started":"2022-06-14T16:54:08.771215Z","shell.execute_reply":"2022-06-14T16:54:08.775972Z"},"jupyter":{"outputs_hidden":false}}},{"cell_type":"code","source":"cum_expl_var_ratio = np.cumsum(ipca.explained_variance_ratio_)","metadata":{"_uuid":"ee1e3ab5-ff09-4688-9210-7ffa62bd8edc","_cell_guid":"bc3a08b6-b16f-49ed-a5e8-3a1ff3add274","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:58.048630Z","iopub.execute_input":"2022-06-14T17:04:58.049689Z","iopub.status.idle":"2022-06-14T17:04:58.055382Z","shell.execute_reply.started":"2022-06-14T17:04:58.049634Z","shell.execute_reply":"2022-06-14T17:04:58.054438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"threshold = 0.99  # Define cumulative variance ratio threshold ","metadata":{"execution":{"iopub.status.busy":"2022-06-14T17:04:58.057051Z","iopub.execute_input":"2022-06-14T17:04:58.057745Z","iopub.status.idle":"2022-06-14T17:04:58.067748Z","shell.execute_reply.started":"2022-06-14T17:04:58.057696Z","shell.execute_reply":"2022-06-14T17:04:58.066799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the cumulative variance\ncum_var = np.cumsum(ipca.explained_variance_ratio_)\n\n# Calculate how many PCs explain 95% of the variance?\nk = np.argmax(cum_var>threshold)\nprint(f'Number of components explaining {threshold:.0%} variance: {k}')\nprint('\\n')\n\nplt.figure(figsize=[10,5])\nplt.title('Cumulative Explained Variance explained by the components')\nplt.ylabel('Cumulative Explained Variance')\nplt.xlabel('Principal components')\nplt.axvline(x=k, color=\"k\", linestyle=\"--\")\nplt.axhline(y=threshold, color=\"r\", linestyle=\"--\")\nax = plt.plot(cum_var)","metadata":{"_uuid":"270889cf-a158-4997-85de-f4b48c59098e","_cell_guid":"0d1d8619-3f08-43b4-8c48-92e1bc85df15","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:58.070009Z","iopub.execute_input":"2022-06-14T17:04:58.070649Z","iopub.status.idle":"2022-06-14T17:04:58.361326Z","shell.execute_reply.started":"2022-06-14T17:04:58.070603Z","shell.execute_reply":"2022-06-14T17:04:58.360496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 4))\nplt.bar(range(k), ipca.explained_variance_ratio_[:k], alpha=0.5, align='center',\n        label='individual explained variance')\nplt.ylabel('Explained variance ratio')\nplt.xlabel('Principal components')\nplt.legend(loc='best')\nplt.tight_layout()","metadata":{"_uuid":"8e0f2b51-4949-4fa6-972f-df1280d8a443","_cell_guid":"7d62143b-a2b7-4eab-a36b-b001d79fe816","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2022-06-14T17:04:58.362583Z","iopub.execute_input":"2022-06-14T17:04:58.363183Z","iopub.status.idle":"2022-06-14T17:04:58.799143Z","shell.execute_reply.started":"2022-06-14T17:04:58.363151Z","shell.execute_reply":"2022-06-14T17:04:58.798225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Save Incremental PCA","metadata":{}},{"cell_type":"code","source":"pk.dump(ipca, open(\"pca.pkl\",\"wb\"))","metadata":{"execution":{"iopub.status.busy":"2022-06-14T17:04:58.800199Z","iopub.execute_input":"2022-06-14T17:04:58.800529Z","iopub.status.idle":"2022-06-14T17:04:58.805916Z","shell.execute_reply.started":"2022-06-14T17:04:58.800500Z","shell.execute_reply":"2022-06-14T17:04:58.805326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}