{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":67356,"databundleVersionId":8006601,"sourceType":"competition"},{"sourceId":8526671,"sourceType":"datasetVersion","datasetId":5088393}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport torch\nfrom itertools import zip_longest\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:21:00.274575Z","iopub.execute_input":"2024-06-07T06:21:00.274999Z","iopub.status.idle":"2024-06-07T06:21:03.649924Z","shell.execute_reply.started":"2024-06-07T06:21:00.274968Z","shell.execute_reply":"2024-06-07T06:21:03.648792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install prince","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:18:12.407438Z","iopub.execute_input":"2024-06-07T06:18:12.407846Z","iopub.status.idle":"2024-06-07T06:18:30.105221Z","shell.execute_reply.started":"2024-06-07T06:18:12.407815Z","shell.execute_reply":"2024-06-07T06:18:30.103624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# !!! I made a mistake on dimensionality reduction. PCA is used for quantitative analysis. For qualitative analysis, we need to consider MCA. so I will correct this analysis.  \n\nCorespondence anlysis is based on principal component analysis. Correspondence analysis is used for categorical variables. If you have more than two categorical variables, you can use multiple correspondence analysis (MCA). \n\nIf you have quantitative data and your dataset is large, Incremental PCA is a good approch.\n","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport pandas as pd\n# Define chunk size and number of components for PCA\nchunk_size = 10000\nn_components = 10\n\n\ntrain_json_file_path = '/kaggle/input/leash-train/leash_train_526.json'\ntest_json_file_path = '/kaggle/input/leash-train/leash_test527.json'\n\n\n\n# Function to load and combine data chunks\ndef load_and_combine_chunks(train_path, test_path, chunk_size):\n    train_chunks = pd.read_json(train_path, orient='records', lines=True, chunksize=chunk_size)\n    test_chunks = pd.read_json(test_path, orient='records', lines=True, chunksize=chunk_size)\n    \n    for train_chunk, test_chunk in zip_longest(train_chunks, test_chunks):\n        if train_chunk is None:\n            combined_chunk = test_chunk\n        elif test_chunk is None:\n            combined_chunk = train_chunk\n        else:\n            combined_chunk = pd.concat([train_chunk, test_chunk], ignore_index=True)\n        \n        yield combined_chunk\n\n\nimport prince\nmca = prince.MCA(\n    n_components=100,\n    n_iter=3,\n    copy=True,\n    check_input=True,\n    engine='sklearn',\n    random_state=42\n)\n\n\nX_all = pd.DataFrame()\nfor combined_chunk in load_and_combine_chunks(train_json_file_path, test_json_file_path, chunk_size):\n    \n    \n    X_combined_chunk = pd.DataFrame(combined_chunk['ecfp'].tolist())\n    X_all = pd.concat([X_all, X_combined_chunk], ignore_index=True)\n    del X_combined_chunk\n\n    \nmca = mca.fit(X_all)\n    \nmca.eigenvalues_summary\n","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:39:14.069482Z","iopub.execute_input":"2024-06-07T06:39:14.070368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mca.eigenvalues_summary\n#mca.transform(X_all)","metadata":{"execution":{"iopub.status.busy":"2024-06-07T06:34:53.787970Z","iopub.execute_input":"2024-06-07T06:34:53.788441Z","iopub.status.idle":"2024-06-07T06:34:53.807176Z","shell.execute_reply.started":"2024-06-07T06:34:53.788395Z","shell.execute_reply":"2024-06-07T06:34:53.805636Z"},"trusted":true},"execution_count":null,"outputs":[]}]}