{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":33317,"databundleVersionId":3293860,"sourceType":"competition"},{"sourceId":12664970,"sourceType":"datasetVersion","datasetId":7985088}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install google-generativeai","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"You will need your developer API key from google studio.\ngemini 2.5 Flash is free. This is the teacher model. ","metadata":{"execution":{"iopub.status.busy":"2025-08-01T06:19:31.976964Z","iopub.execute_input":"2025-08-01T06:19:31.977256Z","iopub.status.idle":"2025-08-01T06:19:31.983029Z","shell.execute_reply.started":"2025-08-01T06:19:31.977233Z","shell.execute_reply":"2025-08-01T06:19:31.981954Z"}}},{"cell_type":"code","source":"import os\nos.environ['GOOGLE_API_KEY'] = ''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#cat /kaggle/input/geoclef2022/get_geoclef2022_datapos.py","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here we take the GeoLifeCleff 20202 dataset and get a small subset (TARGET_SPECIES_ID = 5624)\nThis is the American Black Bear\nFirst We get the positive samples\nInclude is Reverse Geocoding to extract location information including if it is with Protected Area\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport sys\nimport os\nimport pprint\nimport time\nfrom geopy.geocoders import Nominatim\n\n# --- 1. SETUP & INITIALIZATION ---\n\nDATA_PATH = Path('/kaggle/input/geolifeclef-2022-lifeclef-2022-fgvc9/')\n\nif not os.path.isdir('GLC'):\n    print(\"Cloning the GLC repository for helper functions...\")\n    !rm -rf GLC\n    !git clone https://github.com/maximiliense/GLC\nelse:\n    print(\"GLC repository already exists.\")\n\nsys.path.insert(0, '/kaggle/working/GLC')\nfrom GLC._old.data_loading.common import load_patch\n\nprint(\"\\n--- Setup Complete ---\\n\")\n\n# Initialize the geocoder and cache\ngeolocator = Nominatim(user_agent=\"geolifeclef_finetuning_agent_v2\")\nlocation_cache = {}\n\n# --- 2. LOAD DATA & CREATE MAPPINGS ---\n# (This section is unchanged)\nprint(\"Loading data files...\")\ntry:\n    df_obs_us = pd.read_csv(DATA_PATH / \"observations\" / \"observations_us_train.csv\", sep=\";\", index_col=\"observation_id\")\n    df_env = pd.read_csv(DATA_PATH / \"pre-extracted\" / \"environmental_vectors.csv\", sep=\";\", index_col=\"observation_id\")\n    df_landcover_alignment = pd.read_csv(DATA_PATH / \"metadata\" / \"landcover_suggested_alignment.csv\", sep=\";\")\n    print(\"Data loaded successfully.\")\nexcept FileNotFoundError as e:\n    print(f\"Error loading data: {e}\")\n    exit()\n\nalignment_map = df_landcover_alignment.set_index('landcover_code')['suggested_landcover_code'].to_dict()\nmax_code = df_landcover_alignment['landcover_code'].max()\nlandcover_mapping_array = np.zeros(max_code + 1, dtype=np.int16)\nfor original_code, suggested_code in alignment_map.items():\n    landcover_mapping_array[original_code] = suggested_code\nlandcover_label_map = df_landcover_alignment.set_index('suggested_landcover_code')['suggested_landcover_label'].to_dict()\n\n\ndef find_protected_area(address_dict):\n    \"\"\"\n    Searches a geopy address dictionary for common protected area keys.\n    Returns the name of the area if found, otherwise None.\n    \"\"\"\n    # Prioritized list of keys to check in the address dictionary\n    # OSM tagging can be inconsistent, so we check multiple common tags.\n    area_keys = ['national_park', 'forest', 'state_park', 'park', 'nature_reserve', 'protected_area']\n    for key in area_keys:\n        if key in address_dict:\n            return address_dict[key]\n    return None\n\n# --- 4. PROCESS ALL DATA POINTS ---\n\nTARGET_SPECIES_ID = 5624\nTARGET_SPECIES_NAME = \"Ursus americanus (American Black Bear)\"\npresence_observations = df_obs_us[df_obs_us['species_id'] == TARGET_SPECIES_ID]\nprint(f\"\\nFound {len(presence_observations)} presence observations for '{TARGET_SPECIES_NAME}'.\")\nprint(\"Processing all data points with reverse geocoding...\")\n\nall_data_points = []\nprocessed_count = 0\n\nfor observation_id, row in presence_observations.iterrows():\n    try:\n        latitude = row['latitude']\n        longitude = row['longitude']\n\n        location_key = (round(latitude, 2), round(longitude, 2))\n\n        if location_key not in location_cache:\n            print(f\"  ...calling API for {location_key}...\")\n            location = geolocator.reverse(location_key, language='en')\n            time.sleep(1.1) # Be respectful of the API rate limit (1 req/sec)\n\n            if location:\n                address = location.raw.get('address', {})\n                # Use the helper function to check for parks/forests\n                protected_area_name = find_protected_area(address)\n\n                location_cache[location_key] = {\n                    'protected_area': protected_area_name, # New field!\n                    'county': address.get('county'),\n                    'state': address.get('state'),\n                    'country': address.get('country_code', '').upper()\n                }\n            else:\n                location_cache[location_key] = None\n\n        location_context = location_cache[location_key]\n        if not location_context: continue # Skip if geocoding failed\n\n\n        _, _, altitude_patch, aligned_landcover_patch = load_patch(observation_id, DATA_PATH, landcover_mapping=landcover_mapping_array)\n        center_y, center_x = altitude_patch.shape[0] // 2, altitude_patch.shape[1] // 2\n        elevation = altitude_patch[center_y, center_x]\n        env_vector = df_env.loc[observation_id]\n        mean_temp = env_vector['bio_1']\n        annual_precip = env_vector['bio_12']\n        codes, counts = np.unique(aligned_landcover_patch, return_counts=True)\n        total_pixels = aligned_landcover_patch.size\n        landcover_composition_dict = {}\n        percentages = {landcover_label_map.get(c, \"Unknown Code\"): (v / total_pixels) * 100 for c, v in zip(codes, counts)}\n        sorted_percentages = sorted(percentages.items(), key=lambda item: item[1], reverse=True)\n        for label, percent in sorted_percentages:\n            if label != \"No data\" and percent >= 1:\n                landcover_composition_dict[label] = round(percent)\n\n        # Assemble the final dictionary\n        data_point_dict = {\n            'observation_id': observation_id,\n            'location_context': location_context,\n            'latitude': round(latitude, 4),\n            'longitude': round(longitude, 4),\n            'elevation_m': int(elevation),\n            'landcover_composition_percent': landcover_composition_dict,\n            'mean_annual_temp_c': round(mean_temp, 1),\n            'annual_precip_mm': int(annual_precip)\n        }\n\n        all_data_points.append(data_point_dict)\n        processed_count += 1\n\n    except Exception as e:\n        print(f\"\\nSkipping Observation ID {observation_id} due to error: {e}\")\n\nprint(f\"\\nProcessing complete. Successfully created {len(all_data_points)} data points.\")\nprint(\"-\" * 70)\nprint(\"Here is a sample of the output, now with a 'protected_area' field:\\n\")\npprint.pprint(all_data_points[:4])\nimport json\njson.dump(all_data_points, open('american_bear_pos.json', 'w'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Similary for the same species we get the negative samples","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport sys\nimport os\nimport pprint\nfrom tqdm.notebook import tqdm # For a nice progress bar\nimport time\nfrom geopy.geocoders import Nominatim\n\n# --- 1. SETUP & INITIALIZATION ---\n\nDATA_PATH = Path('/kaggle/input/geolifeclef-2022-lifeclef-2022-fgvc9/')\n\nif not os.path.isdir('GLC'):\n    print(\"Cloning the GLC repository for helper functions...\")\n    !rm -rf GLC\n    !git clone https://github.com/maximiliense/GLC\nsys.path.insert(0, '/kaggle/working/GLC')\nfrom GLC._old.data_loading.common import load_patch\n\nprint(\"\\n--- Setup Complete ---\\n\")\n\n# --- 2. LOAD DATA, INITIALIZE GEOCODER & CACHE ---\n\nprint(\"Loading data files...\")\ntry:\n    df_obs_us = pd.read_csv(DATA_PATH / \"observations\" / \"observations_us_train.csv\", sep=\";\", index_col=\"observation_id\")\n    df_obs_fr = pd.read_csv(DATA_PATH / \"observations\" / \"observations_fr_train.csv\", sep=\";\", index_col=\"observation_id\")\n    df_env = pd.read_csv(DATA_PATH / \"pre-extracted\" / \"environmental_vectors.csv\", sep=\";\", index_col=\"observation_id\")\n    df_landcover_alignment = pd.read_csv(DATA_PATH / \"metadata\" / \"landcover_suggested_alignment.csv\", sep=\";\")\n    print(\"Data loaded successfully.\")\nexcept FileNotFoundError as e:\n    print(f\"Error loading data: {e}\")\n    exit()\n\n# Initialize the geocoder and a cache for storing results\ngeolocator = Nominatim(user_agent=\"geolifeclef_finetuning_agent_final\")\nlocation_cache = {}\n\n# --- 3. HELPER FUNCTIONS & MAPPINGS ---\n\ndef find_protected_area(address_dict):\n    \"\"\"Searches a geopy address dictionary for common protected area keys.\"\"\"\n    area_keys = ['national_park', 'forest', 'state_park', 'park', 'nature_reserve', 'protected_area']\n    for key in area_keys:\n        if key in address_dict:\n            return address_dict[key]\n    return None\n\n# Create Landcover Mappings using the correct column names\nalignment_map = df_landcover_alignment.set_index('landcover_code')['suggested_landcover_code'].to_dict()\nmax_code = df_landcover_alignment['landcover_code'].max()\nlandcover_mapping_array = np.zeros(max_code + 1, dtype=np.int16)\nfor original_code, suggested_code in alignment_map.items():\n    landcover_mapping_array[original_code] = suggested_code\nlandcover_label_map = df_landcover_alignment.set_index('suggested_landcover_code')['suggested_landcover_label'].to_dict()\nlandcover_name_to_code_map = {v: k for k, v in landcover_label_map.items()}\n\n# --- 4. SAMPLING PLANS TO GATHER OBSERVATION IDs ---\n\nprint(\"Executing sampling plans to gather negative observation IDs...\")\n\n# Plan A: Biogeographical Mismatch\nneeded_a = 124\nids_plan_a = []\nforest_codes = {landcover_name_to_code_map.get('Broad-leaved Forest'), landcover_name_to_code_map.get('Coniferous Forest')}\nfor obs_id in tqdm(df_obs_fr.index.to_series().sample(frac=1), desc=\"Plan A (France Forests)\"):\n    if needed_a == 0: break\n    try:\n        _, _, _, patch = load_patch(obs_id, DATA_PATH, landcover_mapping=landcover_mapping_array)\n        unique_codes, counts = np.unique(patch, return_counts=True)\n        forest_pixels = sum(counts[np.isin(unique_codes, list(forest_codes))])\n        if (forest_pixels / patch.size) > 0.5: ids_plan_a.append(obs_id); needed_a -= 1\n    except Exception: continue\nprint(f\"  Plan A gathered {len(ids_plan_a)} IDs.\")\n\n# Plan B: Macro-Ecological Mismatch\nneeded_b = 80\nids_plan_b = []\nunsuitable_codes = {landcover_name_to_code_map.get('Natural Grassland/Herbaceous'), landcover_name_to_code_map.get('Cultivated Crops'), landcover_name_to_code_map.get('Pasture/Hay/Intensive Grasslands'), landcover_name_to_code_map.get('Barren Land (Rock/Sand/Clay)')}\nfor obs_id in tqdm(df_obs_us.index.to_series().sample(frac=1), desc=\"Plan B (US Non-Forest)\"):\n    if needed_b == 0: break\n    try:\n        _, _, _, patch = load_patch(obs_id, DATA_PATH, landcover_mapping=landcover_mapping_array)\n        unique_codes, counts = np.unique(patch, return_counts=True)\n        unsuitable_pixels = sum(counts[np.isin(unique_codes, list(unsuitable_codes))])\n        if (unsuitable_pixels / patch.size) > 0.5: ids_plan_b.append(obs_id); needed_b -= 1\n    except Exception: continue\nprint(f\"  Plan B gathered {len(ids_plan_b)} IDs.\")\n\n# Plan C: Local Ecological Contrast\ngrizzly_obs = df_obs_us[df_obs_us['species_id'] == 5626]\nids_plan_c = grizzly_obs.sample(n=min(70, len(grizzly_obs)), random_state=42).index.tolist()\nprint(f\"  Plan C gathered {len(ids_plan_c)} IDs.\")\n\n# Plan D: Anthropogenic Exclusion\nneeded_d = 60\nids_plan_d = []\nurban_codes = {landcover_name_to_code_map.get('Developed High Intensity'), landcover_name_to_code_map.get('Developed Low Intensity')}\nfor obs_id in tqdm(df_obs_us.index.to_series().sample(frac=1), desc=\"Plan D (US Urban)\"):\n    if needed_d == 0: break\n    try:\n        _, _, _, patch = load_patch(obs_id, DATA_PATH, landcover_mapping=landcover_mapping_array)\n        unique_codes, counts = np.unique(patch, return_counts=True)\n        urban_pixels = sum(counts[np.isin(unique_codes, list(urban_codes))])\n        if (urban_pixels / patch.size) > 0.5: ids_plan_d.append(obs_id); needed_d -= 1\n    except Exception: continue\nprint(f\"  Plan D gathered {len(ids_plan_d)} IDs.\")\n\nall_negative_ids = ids_plan_a + ids_plan_b + ids_plan_c + ids_plan_d\nprint(f\"\\nTotal negative sample IDs gathered: {len(all_negative_ids)}. Now processing with geocoding...\")\n\n# --- 5. MAIN PROCESSING LOOP ---\n\nnegative_data_points = []\nfor observation_id in tqdm(all_negative_ids, desc=\"Creating Final Dictionaries\"):\n    try:\n        is_french = str(observation_id).startswith('1')\n        df_obs_source = df_obs_fr if is_french else df_obs_us\n\n        row = df_obs_source.loc[observation_id]\n        latitude = row['latitude']\n        longitude = row['longitude']\n\n        # Perform Reverse Geocoding with Caching\n        location_key = (round(latitude, 2), round(longitude, 2))\n        if location_key not in location_cache:\n            location = geolocator.reverse(location_key, language='en')\n            time.sleep(1.1)\n            if location:\n                address = location.raw.get('address', {})\n                protected_area_name = find_protected_area(address)\n                location_cache[location_key] = {'protected_area': protected_area_name, 'county': address.get('county'), 'state': address.get('state'), 'country': address.get('country_code', '').upper()}\n            else:\n                location_cache[location_key] = None\n\n        location_context = location_cache[location_key]\n        if not location_context: continue\n\n        # Load Patches and Extract Data\n        _, _, altitude_patch, aligned_landcover_patch = load_patch(observation_id, DATA_PATH, landcover_mapping=landcover_mapping_array)\n        elevation = altitude_patch[altitude_patch.shape[0] // 2, altitude_patch.shape[1] // 2]\n        env_vector = df_env.loc[observation_id]\n        mean_temp = env_vector['bio_1']\n        annual_precip = env_vector['bio_12']\n\n        # Process Landcover\n        codes, counts = np.unique(aligned_landcover_patch, return_counts=True)\n        landcover_composition_dict = {}\n        percentages = {landcover_label_map.get(c, \"Unknown Code\"): (v / aligned_landcover_patch.size) * 100 for c, v in zip(codes, counts)}\n        sorted_percentages = sorted(percentages.items(), key=lambda item: item[1], reverse=True)\n        for label, percent in sorted_percentages:\n            if label not in [\"No data\", \"Missing Data\"] and percent >= 1: landcover_composition_dict[label] = round(percent)\n\n        # Assemble the Final Dictionary\n        data_point_dict = {\n            'observation_id': observation_id,\n            'location_context': location_context,\n            'latitude': round(latitude, 4),\n            'longitude': round(longitude, 4),\n            'elevation_m': int(elevation),\n            'landcover_composition_percent': landcover_composition_dict,\n            'mean_annual_temp_c': round(mean_temp, 1),\n            'annual_precip_mm': int(annual_precip) if not np.isnan(annual_precip) else 0\n        }\n\n        negative_data_points.append(data_point_dict)\n\n    except Exception as e:\n        print(f\"\\nSkipping Observation ID {observation_id} due to error: {e}\")\n\n# --- 6. FINAL OUTPUT ---\nprint(f\"\\nProcessing complete. Successfully created {len(negative_data_points)} negative data points.\")\nprint(\"-\" * 70)\nif negative_data_points:\n    print(\"Here is a sample of the first 3 negative data points with full location context:\\n\")\n    pprint.pprint(negative_data_points[:3])\n    import json\n    json.dump(negative_data_points, open('american_bear_neg.json', 'w'))\nelse:\n    print(\"No negative data points were generated. Please check the sampling logic and data paths.\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"positive_data_points = json.load(open('/kaggle/working/american_bear_pos.json'))\nnegative_data_points = json.load(open('/kaggle/working/american_bear_neg.json'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(positive_data_points), len(negative_data_points)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_combined_points = positive_data_points + negative_data_points\nlen(all_combined_points)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\njson.dump(all_combined_points, open('/kaggle/working/american_bear_revcoded.json', 'w'))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"DO the following multiple times till done , or in multiple session to avoid\nrate limit (250 request per day on gemini Flash 2.5)","metadata":{"execution":{"iopub.status.busy":"2025-08-04T04:40:08.183094Z","iopub.execute_input":"2025-08-04T04:40:08.183523Z","iopub.status.idle":"2025-08-04T04:40:08.191743Z","shell.execute_reply.started":"2025-08-04T04:40:08.18349Z","shell.execute_reply":"2025-08-04T04:40:08.190164Z"}}},{"cell_type":"code","source":"import google.generativeai as genai\nimport pandas as pd\nimport numpy as np\nfrom pathlib import Path\nimport sys\nimport os\nimport pprint\nimport time\nimport json\nimport random\n\n# --- 1. CONFIGURE API KEY & BATCH PARAMETERS ---\napi_key = os.getenv(\"GOOGLE_API_KEY\")\ngenai.configure(api_key=api_key)\n\n# Set the number of items to process in one run.\n# You can adjust this based on your daily API limits.\nBATCH_SIZE = 100\nOUTPUT_FILE_NAME = \"finetuning_dataset_complete.jsonl\"\n\n\n# --- 2. DEFINE PROMPT COMPONENTS ---\n\nINSTRUCTION = \"\"\"Analyze the provided location data against the species' known habitat requirements. Predict if the species is likely to be present and provide a concise justification for your reasoning. The justification should directly compare the location's features with the species' needs. Conclude with the prediction on a new line.\"\"\"\n\nECOLOGICAL_DESCRIPTION = \"\"\"\n**Species Profile: Ursus americanus (American Black Bear)**\nA highly adaptable omnivore, its presence is dictated by the availability of habitat that provides both food and protective cover.\n**1. Primary Habitat Requirements:**\n- **Core Need:** Prefers forested areas with dense understory for travel, concealment, and denning.\n- **Landcover Types:** Thrives in deciduous, coniferous, and mixed woodlands, as well as mountains and swamps.\n- **Edge Habitats:** Actively forages where forests meet shrublands, meadows, or riverbanks. Avoids large, open plains or deserts.\n**2. Dietary Needs and Influence on Habitat:**\n- **Driver:** Diet changes seasonally, driving movement.\n- **Seasonal Food:** Needs emergent vegetation (spring), berries/insects (summer), and hard mast like acorns/nuts (autumn) for hibernation.\n**3. Climate and Elevation Tolerance:**\n- **Climate Adaptability:** Wide tolerance from subtropical to subarctic.\n- **Critical Factor:** Requires a climate with a cold period sufficient for hibernation.\n- **Elevation Range:** Found from sea level to over 3,000 meters.\n**4. Geographic Range and Human Interaction:**\n- **Human Factor:** Presence within a National Park or National Forest is a strong positive indicator of suitable habitat.\n\"\"\"\n\n# --- 3. HELPER FUNCTION TO CREATE THE TEACHER PROMPT ---\n\ndef create_teacher_prompt(data_point):\n    \"\"\"Formats a data point and its ground-truth label into a full prompt for the teacher LLM.\"\"\"\n    \n    # Format location context\n    loc_ctx = data_point['location_context']\n    location_str = \"- Geographic Area: Geocoding failed or not available.\"\n    if loc_ctx:\n        protected_area = loc_ctx.get('protected_area') or \"Not specified\"\n        county = loc_ctx.get('county') or \"N/A\"\n        state = loc_ctx.get('state') or \"N/A\"\n        location_str = f\"- Geographic Area: {protected_area}\\n- Civic Location: {county}, {state}, {loc_ctx.get('country', 'N/A')}\"\n\n    # Format landcover composition\n    landcover_dict = data_point['landcover_composition_percent']\n    landcover_str = \", \".join([f\"{k} ({v}%)\" for k, v in landcover_dict.items()]) if landcover_dict else \"Undetermined\"\n        \n    # This is the \"input\" part of our final dataset\n    input_data_str = (\n        f\"**Location Data:**\\n\"\n        f\"{location_str}\\n\"\n        f\"- Coordinates: {data_point['latitude']}, {data_point['longitude']}\\n\"\n        f\"- Elevation: {data_point['elevation_m']}m\\n\"\n        f\"- Landcover Composition: {landcover_str}\\n\"\n        f\"- Climate: Mean Annual Temp {data_point['mean_annual_temp_c']}°C, Annual Precip {data_point['annual_precip_mm']}mm\\n\\n\"\n        f\"**Species Information:**\\n{ECOLOGICAL_DESCRIPTION}\"\n    )\n\n    # This is the full prompt for the teacher model, telling it the correct answer\n    ground_truth_label = data_point['label']\n    teacher_prompt = (\n        f\"You are an expert ecologist. The ground truth is that the species is **{ground_truth_label}** at this location.\\n\"\n        f\"Your task is to write a concise justification explaining *why* this is the case by comparing the location's features with the species' needs.\\n\"\n        f\"Conclude with the prediction on a new line: 'Prediction: {ground_truth_label}'\\n\\n\"\n        f\"--- INPUT DATA ---\\n{input_data_str}\"\n    )\n    \n    return teacher_prompt, input_data_str\n\n\n# --- 4. PREPARE THE DATA AND MANAGE STATE ---\n\n# This assumes you have 'positive_data_points' and 'negative_data_points' lists ready.\n# If you don't, load them from files or generate them first.\n# For demonstration, we'll create dummy lists. Replace these with your real data.\n# positive_data_points = [...] \n# negative_data_points = [...]\n\n# Add labels to each dictionary\nfor point in positive_data_points: point['label'] = 'Present'\nfor point in negative_data_points: point['label'] = 'Absent'\n\n# Combine and shuffle the data for balanced processing\nall_combined_points = positive_data_points + negative_data_points\nrandom.shuffle(all_combined_points)\n\n# State management: Check for already processed items\nprocessed_ids = set()\nif os.path.exists(OUTPUT_FILE_NAME):\n    with open(OUTPUT_FILE_NAME, 'r') as f:\n        for line in f:\n            # We add the original observation ID from the input string to our set\n            # This is a bit complex, but robust.\n            try:\n                # Assuming 'observation_id' is stored in the 'input' string after 'Coordinates:'\n                # This part is a placeholder; a better way would be to save the ID in the JSONL.\n                # Let's adjust the final JSONL to include the ID for easier state management.\n                data = json.loads(line)\n                processed_ids.add(data['observation_id'])\n            except (json.JSONDecodeError, KeyError):\n                continue\n\npoints_to_process = [p for p in all_combined_points if p['observation_id'] not in processed_ids]\n\nprint(f\"\\nFound {len(all_combined_points)} total data points.\")\nprint(f\"Found {len(processed_ids)} already processed items in '{OUTPUT_FILE_NAME}'.\")\nprint(f\"{len(points_to_process)} items remaining to process.\")\n\n\n# --- 5. GENERATE THE FINETUNING DATASET IN BATCHES ---\n\nmodel = genai.GenerativeModel(model_name='gemini-2.5-flash')\nitems_in_this_batch = points_to_process[:BATCH_SIZE]\nnewly_processed_count = 0\n\nprint(f\"\\nStarting new batch. Attempting to process {len(items_in_this_batch)} items...\")\n\nif not items_in_this_batch:\n    print(\"No new items to process. You're all done!\")\nelse:\n    with open(OUTPUT_FILE_NAME, 'a') as f:\n        for i, data_point in enumerate(items_in_this_batch):\n            teacher_prompt, input_data_for_saving = create_teacher_prompt(data_point)\n            \n            try:\n                print(f\"  Processing batch item {i+1}/{len(items_in_this_batch)} (ID: {data_point['observation_id']}, Label: {data_point['label']})...\")\n                response = model.generate_content(teacher_prompt)\n                \n                if not response.parts:\n                    print(f\"    -> WARNING: Received an empty/blocked response. Skipping.\")\n                    continue\n                \n                generated_output = response.text.strip()\n                \n                finetuning_example = {\n                    \"instruction\": INSTRUCTION,\n                    \"input\": input_data_for_saving, # Save the clean input\n                    \"output\": generated_output,\n                    \"observation_id\": data_point['observation_id'] # Save ID for state management\n                }\n                \n                # Append the result to the file immediately\n                json.dump(finetuning_example, f)\n                f.write('\\n')\n                newly_processed_count += 1\n\n                time.sleep(1.2) # Respect the 60 requests/minute limit\n\n            except Exception as e:\n                print(f\"    -> ERROR: {e}. Skipping item.\")\n                # Optional: Add a longer sleep after an error\n                time.sleep(5)\n\n    print(\"\\n--- Batch Complete ---\")\n    print(f\"Successfully processed and saved {newly_processed_count} new examples to '{OUTPUT_FILE_NAME}'.\")\n    print(\"Run this script again tomorrow to process the next batch.\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}