{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":83481,"databundleVersionId":9399961,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport os\nfrom rich.console import Console\nfrom rich.tree import Tree\n\ndef generate_tree(directory):\n    tree = Tree(directory)\n\n    def add_to_tree(path, parent):\n        for entry in sorted(os.listdir(path)):\n            full_path = os.path.join(path, entry)\n            if os.path.isdir(full_path):\n                branch = parent.add(f\"[bold cyan]{entry}[/]\")\n                add_to_tree(full_path, branch)\n            else:\n                parent.add(f\"[green]{entry}[/]\")\n\n    add_to_tree(directory, tree)\n    return tree\n\n# Set up console\nconsole = Console()\n\n# Define the directory you want to visualize\ndirectory = \"/kaggle/input\"  # Change this path as needed\n\n# Generate the directory tree\ntree = generate_tree(directory)\n\n# Print the directory tree\nconsole.print(tree)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-26T04:14:13.496858Z","iopub.execute_input":"2024-08-26T04:14:13.497403Z","iopub.status.idle":"2024-08-26T04:14:14.575182Z","shell.execute_reply.started":"2024-08-26T04:14:13.497346Z","shell.execute_reply":"2024-08-26T04:14:14.573886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Baselines\n\nGenerate some simple baseline submissions for the leaderboard","metadata":{}},{"cell_type":"code","source":"!pip install duckdb --quiet\nimport duckdb\nprint(duckdb.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:14:14.577901Z","iopub.execute_input":"2024-08-26T04:14:14.578666Z","iopub.status.idle":"2024-08-26T04:14:33.797394Z","shell.execute_reply.started":"2024-08-26T04:14:14.578609Z","shell.execute_reply":"2024-08-26T04:14:33.796218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nimport duckdb\nimport pandas as pd\nimport numpy as np\n\n# Suppress specific DeprecationWarning\n#warnings.filterwarnings(\"ignore\", category=DeprecationWarning)\n\n# Connect to DuckDB in memory\ncon = duckdb.connect()\n\n# Load the test data\ncmd = \"\"\"\n    SELECT \n        beacon.report_pubKey AS pubKey,\n        witness.report_signal AS signal,\n        witness.latitude AS witness_latitude,\n        witness.longitude AS witness_longitude\n    FROM read_parquet('/kaggle/input/radio-direction-finding-challenge/test_beacons.parquet/*/*.parquet')\n\"\"\"\ntest_df = con.execute(cmd).fetchdf()\n\n# Define a function to calculate normalized confidence based on signal strength\ndef calculate_confidence(group):\n    mean_signal = group['signal'].mean()\n    max_signal = test_df['signal'].max()  # Reference the maximum signal in the entire dataset\n    confidence = 100 * mean_signal / max_signal\n    return confidence\n\n# 1. Strongest RSSI Location\ndef strongest_rssi(group):\n    best_row = group.loc[group['signal'].idxmax()]\n    confidence = 100  # Full confidence for the strongest signal\n    return pd.Series({'latitude': best_row['witness_latitude'], 'longitude': best_row['witness_longitude'], 'confidence': confidence})\n\nstrongest_df = test_df.groupby('pubKey', group_keys=False).apply(strongest_rssi).reset_index()\nstrongest_df.to_csv('strongest_rssi_location.csv', index=False)\n\n# 2. Average of Witness Locations\ndef weighted_avg(group):\n    weights = group['signal']\n    lat = np.average(group['witness_latitude'], weights=weights)\n    lon = np.average(group['witness_longitude'], weights=weights)\n    confidence = calculate_confidence(group)  # Confidence based on average signal strength\n    return pd.Series({'latitude': lat, 'longitude': lon, 'confidence': confidence})\n\nweighted_avg_df = test_df.groupby('pubKey', group_keys=False).apply(weighted_avg).reset_index()\nweighted_avg_df.to_csv('average_witness_location.csv', index=False)\n\n# 3. Median of Witness Locations\ndef median_location(group):\n    lat = group['witness_latitude'].median()\n    lon = group['witness_longitude'].median()\n    confidence = calculate_confidence(group)  # Confidence based on average signal strength\n    return pd.Series({'latitude': lat, 'longitude': lon, 'confidence': confidence})\n\nmedian_df = test_df.groupby('pubKey', group_keys=False).apply(median_location).reset_index()\nmedian_df.to_csv('median_witness_location.csv', index=False)\n\n# 4. Center of Witnesses (Geometric Center)\ndef geometric_center(group):\n    lat = group['witness_latitude'].mean()\n    lon = group['witness_longitude'].mean()\n    confidence = calculate_confidence(group)  # Confidence based on average signal strength\n    return pd.Series({'latitude': lat, 'longitude': lon, 'confidence': confidence})\n\ncenter_df = test_df.groupby('pubKey', group_keys=False).apply(geometric_center).reset_index()\ncenter_df.to_csv('center_of_witnesses.csv', index=False)\n\n# 5. Farthest Witness Location\ndef farthest_witness(group):\n    worst_row = group.loc[group['signal'].idxmin()]\n    confidence = calculate_confidence(group)  # Lower confidence since it's based on weaker signals\n    return pd.Series({'latitude': worst_row['witness_latitude'], 'longitude': worst_row['witness_longitude'], 'confidence': confidence})\n\nfarthest_df = test_df.groupby('pubKey', group_keys=False).apply(farthest_witness).reset_index()\nfarthest_df.to_csv('farthest_witness_location.csv', index=False)\n\n# 6. Random Witness Location\ndef random_witness(group):\n    row = group.sample(n=1)\n    confidence = calculate_confidence(group)  # Arbitrary confidence, could be mid-range\n    return pd.Series({'latitude': row['witness_latitude'].values[0], 'longitude': row['witness_longitude'].values[0], 'confidence': confidence})\n\nrandom_df = test_df.groupby('pubKey', group_keys=False).apply(random_witness).reset_index()\nrandom_df.to_csv('random_witness_location.csv', index=False)\n\nprint(\"All baseline models have been generated and saved as CSV files.\")\n","metadata":{"execution":{"iopub.status.busy":"2024-08-26T04:14:33.799127Z","iopub.execute_input":"2024-08-26T04:14:33.799543Z"},"trusted":true},"execution_count":null,"outputs":[]}]}