{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What is about ?\n\nFeature engineering for the  Kaggle competition https://www.kaggle.com/competitions/open-problems-multimodal\n\nDimensional reductions provide kind of non-linear feature engineering.\nDifferent methods and their different parameters allow to construct plenty features. \n\nProduced files are stored in the current dataset: https://www.kaggle.com/datasets/alexandervc/feature-shop-for-multimodal-singlecell-competition\n\nDue to huge size - we first reduced dimension by PCA/TruncatedSVD to 200, 500 - using \"Saturn\" cloud - which has 100+G RAM. And as the next steps we use dimensional reductions from that prelimanary reduced space - that can be done done on Kaggle (with its 16G RAM).\nThe present script performs that later step.\n\nNames of the files correspond to the method and params  were used to produce it, and what preliminary dim-red (PCA/TruncSVD) was used.\nColumns - just the indices of the dimensions.\n\nAll files contain both features for train and test. First come train samples, later test.\nFor CITEseq part - first 70988 elements - train, and later 48663 - test. Overall 119651 samples.\n\n\nSee also discussion: https://www.kaggle.com/competitions/open-problems-multimodal/discussion/359683\n\n## Versions \n\n\n### 8 cosmetic changes \n\n\n### 7 Trimap \n\n    1 hour \n    Might be also good for outlier detection i.e. models can better see that some samples are outliers with the help of these features\n\n### 6 UMAP and density preserving is turned ON,  with different params \n\n    It might be good for outlier detection  i.e. models can better see that some samples are outliers with the help of these features\n    3 hours 17 minutes.\n    n_dimensions = 100 , 6 param options consered \n\n\n### 5 UMAP with different params \n\n    1.5 hours.\n    n_dimensions = 100 , 6 param options consered \n\n### 4 RandomTreesEmbedding\n\n    Preliminary step - only PCA200\n\n### 3 Factor analysis several variants of components\n\n    Preliminary step - only PCA200\n\n### 2 - ICA dimensional reduction with several components\n\n### 1 - NCVIS dimensional reduction (it is like UMAP but faster), but only to 2-dimensions. \n\n    Several versions - different by preliminary step - PCA, or TruncatedSVD and dimensions for them\n    \n","metadata":{}},{"cell_type":"code","source":"run_Trimap = False \nrun_UMAPdense = False # 3 hours 17 mins \nrun_UMAP = False # 1.5 hour for 6 params configs for 100 dimensions\nrun_RProj = True\nrun_RTrees = True\nrun_FA = True\nrun_ICA = True\nrun_NCVis = True\n\n\n\nimport time\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:23:32.182669Z","iopub.execute_input":"2022-10-31T19:23:32.183100Z","iopub.status.idle":"2022-10-31T19:23:32.190339Z","shell.execute_reply.started":"2022-10-31T19:23:32.183068Z","shell.execute_reply":"2022-10-31T19:23:32.188914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-31T19:00:29.157916Z","iopub.execute_input":"2022-10-31T19:00:29.158821Z","iopub.status.idle":"2022-10-31T19:00:29.182079Z","shell.execute_reply.started":"2022-10-31T19:00:29.158772Z","shell.execute_reply":"2022-10-31T19:00:29.181458Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"соединяем трейн+тест датасеты для 200","metadata":{}},{"cell_type":"code","source":"train_data_200 = pd.read_csv('../input/msci-cite-denoised-pca200/X_denoised_PCA200.csv')\ntest_data_200 = pd.read_csv('../input/msci-cite-denoised-pca200/Xt_denoised_PCA200.csv')\ndata_200 = pd.concat([train_data_200, test_data_200])\ndata_200.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-30T20:36:24.013543Z","iopub.execute_input":"2022-10-30T20:36:24.013849Z","iopub.status.idle":"2022-10-30T20:36:26.746967Z","shell.execute_reply.started":"2022-10-30T20:36:24.013825Z","shell.execute_reply":"2022-10-30T20:36:26.746157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_200.to_csv('msci-cite-denoised_train_and_test_PCA200.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-30T20:36:48.326929Z","iopub.execute_input":"2022-10-30T20:36:48.327238Z","iopub.status.idle":"2022-10-30T20:37:04.056150Z","shell.execute_reply.started":"2022-10-30T20:36:48.327211Z","shell.execute_reply":"2022-10-30T20:37:04.055058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"соединяем трейн+тест датасеты для 500","metadata":{}},{"cell_type":"code","source":"#train_data_500 = pd.read_csv('../input/msci-multiome-train-truncatedsvd/X_TSVD_500.csv')\n#test_data_500 = pd.read_csv('../input/msci-multiome-train-truncatedsvd/Xtest_TSVD_500.csv')\n#data_500 = pd.concat([train_data_500, test_data_500])\n#data_500.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-18T06:38:23.355331Z","iopub.execute_input":"2022-10-18T06:38:23.355753Z","iopub.status.idle":"2022-10-18T06:38:48.065296Z","shell.execute_reply.started":"2022-10-18T06:38:23.355718Z","shell.execute_reply":"2022-10-18T06:38:48.064329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"str_data_inf = 'MAGIC_CITE_'","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:09:57.257745Z","iopub.execute_input":"2022-10-31T19:09:57.259083Z","iopub.status.idle":"2022-10-31T19:09:57.266448Z","shell.execute_reply.started":"2022-10-31T19:09:57.259021Z","shell.execute_reply":"2022-10-31T19:09:57.264440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n#fn = '/kaggle/input/feature-shop-for-multimodal-singlecell-competition/citeseq_train_and_test_PCA200.csv'\n#data_suffix = 'TruncatedSVD200_niter7_rs42'\n#data_suffix = 'PCA500'\ndata_suffix = 'PCA200'\nfn = '../input/feature-shop-magic-pca200/msci-cite-denoised_train_and_test_PCA200.csv'\n\ndf = pd.read_csv(fn, index_col = 0)\n\nprint(df.shape,df.info())\ndisplay(df)\nr = df.values","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:10:18.313392Z","iopub.execute_input":"2022-10-31T19:10:18.313844Z","iopub.status.idle":"2022-10-31T19:10:27.495365Z","shell.execute_reply.started":"2022-10-31T19:10:18.313805Z","shell.execute_reply":"2022-10-31T19:10:27.493970Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# NCVIS (like umap, but faster)","metadata":{}},{"cell_type":"code","source":"!pip install ncvis","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:43:31.703645Z","iopub.execute_input":"2022-10-31T19:43:31.704078Z","iopub.status.idle":"2022-10-31T19:43:43.553411Z","shell.execute_reply.started":"2022-10-31T19:43:31.704046Z","shell.execute_reply":"2022-10-31T19:43:43.551879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif run_NCVis:\n    import time\n    from sklearn.decomposition import FastICA\n    #import trimap\n    import ncvis\n\n    \n\n    for n_components in [2,3,5,10,20,50]:\n        \n        #reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        #reducer = trimap.TRIMAP(n_dims=n_components)# , random_state=0, whiten='unit-variance')\n        reducer = ncvis.NCVis()\n        str_inf = str_data_inf+'NCVISfrom'+data_suffix +'_n_components'+str(n_components) + 'rs0'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:49:25.890916Z","iopub.execute_input":"2022-10-31T19:49:25.892113Z","iopub.status.idle":"2022-10-31T19:49:25.928404Z","shell.execute_reply.started":"2022-10-31T19:49:25.892061Z","shell.execute_reply":"2022-10-31T19:49:25.926867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ICA","metadata":{}},{"cell_type":"code","source":"%%time\nif run_ICA: \n    import time\n    from sklearn.decomposition import FastICA\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n\n    for n_components in [2,3,5,10,20,50]:\n        reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        str_inf = str_data_inf+'ICAfrom'+data_suffix +'_n_components'+str(n_components) + 'rs0'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:20:36.119502Z","iopub.execute_input":"2022-10-31T19:20:36.119976Z","iopub.status.idle":"2022-10-31T19:22:51.343198Z","shell.execute_reply.started":"2022-10-31T19:20:36.119939Z","shell.execute_reply":"2022-10-31T19:22:51.341861Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# FactorAnalysis","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom sklearn.decomposition import FactorAnalysis\n#from sklearn.decomposition import LatentDirichletAllocation\nfrom sklearn.ensemble import RandomTreesEmbedding\nfrom sklearn.random_projection import SparseRandomProjection\n\nif run_FA: \n    import time\n    from sklearn.decomposition import FastICA\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n\n    for n_components in [2,3,5,10,20,50]:\n        #reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        reducer = FactorAnalysis(n_components=n_components, random_state=0)\n        str_inf = str_data_inf+'FAfrom'+data_suffix +'_n_components'+str(n_components) + 'rs0'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:15:36.732138Z","iopub.execute_input":"2022-10-31T19:15:36.732661Z","iopub.status.idle":"2022-10-31T19:16:56.786925Z","shell.execute_reply.started":"2022-10-31T19:15:36.732597Z","shell.execute_reply":"2022-10-31T19:16:56.785471Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RundomTreesEmbedding","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom sklearn.decomposition import FactorAnalysis\n#from sklearn.decomposition import LatentDirichletAllocation\nfrom sklearn.ensemble import RandomTreesEmbedding\nfrom sklearn.random_projection import SparseRandomProjection\nimport scipy.sparse\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.decomposition import TruncatedSVD\n\nif run_RTrees: \n    import time\n    from sklearn.decomposition import FastICA\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n\n    for n_components in [2,3,5,10,20,50]:\n        #reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        #reducer = FactorAnalysis(n_components=n_components, random_state=0)\n        #reducer = RandomTreesEmbedding(n_estimators=n_components, random_state=0, max_depth=5)\n        reducer = make_pipeline(RandomTreesEmbedding(n_estimators=200, random_state=0, max_depth=5), \n                                TruncatedSVD(n_components=n_components) )\n        \n        str_inf = str_data_inf+'RTreesfrom'+data_suffix +'_n_components'+str(n_components) + 'ne200md5rs0'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        if scipy.sparse.issparse(r2):\n            r2 = r2.toarray()\n            \n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:11:39.159008Z","iopub.execute_input":"2022-10-31T19:11:39.159501Z","iopub.status.idle":"2022-10-31T19:13:19.606701Z","shell.execute_reply.started":"2022-10-31T19:11:39.159467Z","shell.execute_reply":"2022-10-31T19:13:19.604958Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# RProj","metadata":{}},{"cell_type":"code","source":"%%time\n\nfrom sklearn.decomposition import FactorAnalysis\n#from sklearn.decomposition import LatentDirichletAllocation\nfrom sklearn.ensemble import RandomTreesEmbedding\nfrom sklearn.random_projection import SparseRandomProjection\nimport scipy.sparse\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.decomposition import TruncatedSVD\n\nif run_RProj: \n    import time\n    from sklearn.decomposition import FastICA\n    import seaborn as sns\n    import matplotlib.pyplot as plt\n\n    for n_components in [2,3,5,10,20,50]:\n        #reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        #reducer = FactorAnalysis(n_components=n_components, random_state=0)\n        #reducer = RandomTreesEmbedding(n_estimators=n_components, random_state=0, max_depth=5)\n        #reducer = make_pipeline(RandomTreesEmbedding(n_estimators=200, random_state=0, max_depth=5), \n        #                        TruncatedSVD(n_components=n_components) )\n        reducer = SparseRandomProjection(n_components=n_components, random_state=42)\n\n        str_inf = str_data_inf+'RProjfrom'+data_suffix +'_n_components'+str(n_components) + 'rs42'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        if scipy.sparse.issparse(r2):\n            r2 = r2.toarray()\n            \n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:10:36.048947Z","iopub.execute_input":"2022-10-31T19:10:36.050330Z","iopub.status.idle":"2022-10-31T19:10:57.309472Z","shell.execute_reply.started":"2022-10-31T19:10:36.050284Z","shell.execute_reply":"2022-10-31T19:10:57.307892Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ","metadata":{}},{"cell_type":"markdown","source":"# UMAP","metadata":{}},{"cell_type":"code","source":"pip install umap-learn","metadata":{"execution":{"iopub.status.busy":"2022-10-31T18:57:07.050291Z","iopub.execute_input":"2022-10-31T18:57:07.050548Z","iopub.status.idle":"2022-10-31T18:57:15.238647Z","shell.execute_reply.started":"2022-10-31T18:57:07.050523Z","shell.execute_reply":"2022-10-31T18:57:15.237301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif run_UMAP:\n    import time\n    import umap\n    for n_components in [2,3,5,10,50,100]:\n        for min_dist in [0.1, 0.9]: # 0.1 - defailt min_dist\n            for n_neighbors in [5,15,100]: # 15 - default n_neighbors,\n                t0 = time.time()\n                print('%.1f seconds passed '%( time.time() - t0) ) \n                reducer = umap.UMAP(n_components = n_components, n_neighbors = n_neighbors, min_dist = min_dist, random_state=42  )# random_state=42 # metric = \n                str_inf = str_data_inf+'UMAPfrom'+data_suffix+'_n_components'+str(n_components) +'_n_neighbors' + str(n_neighbors) +\\\n                    '_min_dist'+str(min_dist).replace('.','d') +  '_rs42'\n                r2 = reducer.fit_transform(r)\n                fn = str_inf + '.csv'\n                df2 = pd.DataFrame(index = df.index, data = r2)\n                print( df2.info() )\n\n                df2.to_csv(fn)\n                display(df2.head(2) )           \n                print(fn)\n                print('%.1f seconds passed '%( time.time() - t0) ) \n                cm = df2.corr() \n                display(cm)\n                plt.figure(figsize = (10,10))\n                sns.heatmap(cm)\n                plt.show()\n                ","metadata":{"execution":{"iopub.status.busy":"2022-10-31T19:00:57.439190Z","iopub.execute_input":"2022-10-31T19:00:57.439546Z","iopub.status.idle":"2022-10-31T19:04:51.731579Z","shell.execute_reply.started":"2022-10-31T19:00:57.439519Z","shell.execute_reply":"2022-10-31T19:04:51.730454Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# UMAP in \"density preserving mode\"","metadata":{}},{"cell_type":"code","source":"%%time\nif run_UMAPdense:\n    import time\n    import umap\n    for n_components in [2,3,5,10,50]: #[2,3,5,10,50,100]:\n        for min_dist in [0.1]: # 0.1 - defailt min_dist\n            for n_neighbors in [15]: # 15 - default n_neighbors,\n                t0 = time.time()\n                print('%.1f seconds passed '%( time.time() - t0) ) \n                reducer = umap.UMAP(densmap=True,n_components = n_components, n_neighbors = n_neighbors, min_dist = min_dist, random_state=42  )# random_state=42 # metric = \n                str_inf = str_data_inf+'UMAPDensefrom'+data_suffix+'_n_components'+str(n_components) +'_n_neighbors' + str(n_neighbors) +\\\n                    '_min_dist'+str(min_dist).replace('.','d') +  '_rs42'\n                r2 = reducer.fit_transform(r)\n                fn = str_inf + '.csv'\n                df2 = pd.DataFrame(index = df.index, data = r2)\n                print( df2.info() )\n\n                df2.to_csv(fn)\n                display(df2.head(2) )           \n                print(fn)\n                print('%.1f seconds passed '%( time.time() - t0) ) \n                cm = df2.corr() \n                display(cm)\n                plt.figure(figsize = (10,10))\n                sns.heatmap(cm)\n                plt.show()\n                ","metadata":{"execution":{"iopub.status.busy":"2022-10-19T03:18:44.475461Z","iopub.execute_input":"2022-10-19T03:18:44.475841Z","iopub.status.idle":"2022-10-19T03:18:44.484694Z","shell.execute_reply.started":"2022-10-19T03:18:44.475812Z","shell.execute_reply":"2022-10-19T03:18:44.483802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Trimap","metadata":{}},{"cell_type":"code","source":"!pip install trimap ","metadata":{"execution":{"iopub.status.busy":"2022-10-19T03:18:05.833102Z","iopub.execute_input":"2022-10-19T03:18:05.833880Z","iopub.status.idle":"2022-10-19T03:18:19.790801Z","shell.execute_reply.started":"2022-10-19T03:18:05.833844Z","shell.execute_reply":"2022-10-19T03:18:19.789427Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nif run_Trimap:\n    import time\n    #from sklearn.decomposition import FastICA\n    import trimap \n    for n_components in [2,3,5,10,20,50]:\n        #reducer = FastICA(n_components=n_components, random_state=0, whiten='unit-variance')\n        reducer = trimap.TRIMAP(n_dims=n_components) #, random_state=0, whiten='unit-variance')\n        str_inf = str_data_inf+'TRIMAPfrom'+data_suffix+'_n_components'+str(n_components) # + 'rs0'\n        t0 = time.time()\n        print(str_inf, '%.1f seconds passed '%( time.time() - t0) ) \n        r2 = reducer.fit_transform(r)\n        fn = str_inf + '.csv'\n        df2 = pd.DataFrame(index = df.index, data = r2)\n        print( df2.info() )\n\n        df2.to_csv(fn)\n        display(df2.head(2) )           \n        print(fn)\n        print('%.1f seconds passed '%( time.time() - t0) ) \n        \n        cm = df2.corr() \n        display(cm)\n        plt.figure(figsize = (10,10))\n        sns.heatmap(cm)\n        plt.show()\n        ","metadata":{"execution":{"iopub.status.busy":"2022-10-19T03:18:34.938558Z","iopub.execute_input":"2022-10-19T03:18:34.939130Z","iopub.status.idle":"2022-10-19T03:18:34.946800Z","shell.execute_reply.started":"2022-10-19T03:18:34.939097Z","shell.execute_reply":"2022-10-19T03:18:34.945561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"2+2","metadata":{"execution":{"iopub.status.busy":"2022-10-13T09:24:02.64155Z","iopub.execute_input":"2022-10-13T09:24:02.642103Z","iopub.status.idle":"2022-10-13T09:24:02.651696Z","shell.execute_reply.started":"2022-10-13T09:24:02.642051Z","shell.execute_reply":"2022-10-13T09:24:02.650185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"2","metadata":{"execution":{"iopub.status.busy":"2022-10-13T09:24:02.653019Z","iopub.execute_input":"2022-10-13T09:24:02.653497Z","iopub.status.idle":"2022-10-13T09:24:02.663884Z","shell.execute_reply.started":"2022-10-13T09:24:02.653446Z","shell.execute_reply":"2022-10-13T09:24:02.662592Z"},"trusted":true},"execution_count":null,"outputs":[]}]}