{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":84493,"databundleVersionId":9849268,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-15T10:06:44.404362Z","iopub.execute_input":"2024-10-15T10:06:44.405238Z","iopub.status.idle":"2024-10-15T10:06:45.597793Z","shell.execute_reply.started":"2024-10-15T10:06:44.405188Z","shell.execute_reply":"2024-10-15T10:06:45.596456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The organizers provided responder_0 to responder_8, along with the responders.csv file. The training dataset also includes a total of 9 responders, but we are only required to predict responder_6.\n\nThis led me to a hypothesis: can the relationships between the responders be analyzed using the responders.csv file? For example, can responder_6 be represented by the other responders?\n\nHere’s what I did (thanks to SLi for the additional insights in the comments：https://www.kaggle.com/code/wjjjjs/what-is-responder-6/comments#3018612):\nI applied one-hot encoding to responders.csv and then used brute-force enumeration to verify if any relationships exist between responder_6 and the other responders.\n\nA big thanks to JCM for validating the expression I derived for r6:\nhttps://www.kaggle.com/code/wjjjjs/what-is-responder-6/comments#3018634  \n**JCM's validation showed that the linear expression I derived does not hold for the actual data, so using these expressions directly doesn’t have much significance.**\n\nThis notebook is meant to share my hypothesis about responders.csv. For now, deeper analysis is still needed. I will continue to update if there are further developments. Thanks to everyone in the comments for the questions and feedback!","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nfrom itertools import combinations, product\n# data from responders.csv\ndata = np.array([\n    [1, 0, 1, 0, 0],\n    [1, 0, 0, 1, 0],\n    [1, 1, 0, 0, 0],\n    [0, 0, 1, 0, 1],\n    [0, 0, 0, 1, 1],\n    [0, 1, 0, 0, 1],\n    [0, 0, 1, 0, 0],\n    [0, 0, 0, 1, 0],\n    [0, 1, 0, 0, 0]\n])\n\ndf = pd.DataFrame(data.T, columns=[f'r{i}' for i in range(9)])\n\ncandidates = [1, 2, 4, 5, 7, 8]  # r1, r2, r4, r5, r7, r8\n\nexpressions = []\n\nfor i in range(1, len(candidates) + 1):\n    for combo in combinations(candidates, i):\n        for signs in product([-1, 1], repeat=len(combo)):\n            expr = \"df['r6'] == df['r0']\"\n            for sign, row in zip(signs, combo):\n                expr += f\" + {sign} * df['r{row}']\"  # df 中的列名\n            expressions.append(expr)\n\nresults = []\nfor expr in expressions:\n    result = eval(expr)\n    if all(result):\n        print(expr)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T10:06:48.935389Z","iopub.execute_input":"2024-10-15T10:06:48.935926Z","iopub.status.idle":"2024-10-15T10:06:49.628502Z","shell.execute_reply.started":"2024-10-15T10:06:48.935877Z","shell.execute_reply":"2024-10-15T10:06:49.627378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidates = [1, 2, 4, 5, 7, 8]  # r1, r2, r4, r5, r7, r8\n\nexpressions = []\n\nfor i in range(1, len(candidates) + 1):\n    for combo in combinations(candidates, i):\n        for signs in product([-1, 1], repeat=len(combo)):\n            expr = \"df['r6'] == df['r3']\"\n            for sign, row in zip(signs, combo):\n                expr += f\" + {sign} * df['r{row}']\"  # df 中的列名\n            expressions.append(expr)\n\nresults = []\nfor expr in expressions:\n    result = eval(expr)\n    if all(result):\n        print(expr)\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-15T10:07:04.111665Z","iopub.execute_input":"2024-10-15T10:07:04.112063Z","iopub.status.idle":"2024-10-15T10:07:04.778249Z","shell.execute_reply.started":"2024-10-15T10:07:04.112024Z","shell.execute_reply":"2024-10-15T10:07:04.777275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO r6(responder_6) can be represented by the rest, so can we predict all the responders and calculate responder_6?","metadata":{},"execution_count":null,"outputs":[]}]}