{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":97984,"databundleVersionId":14096757,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q playwright\n!python -m playwright install chromium > /dev/null 2>&1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:25:49.407954Z","iopub.execute_input":"2026-01-17T01:25:49.408304Z","iopub.status.idle":"2026-01-17T01:26:16.869846Z","shell.execute_reply.started":"2026-01-17T01:25:49.408252Z","shell.execute_reply":"2026-01-17T01:26:16.868709Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from playwright.async_api import async_playwright\nimport time\nimport requests\nfrom bs4 import BeautifulSoup\nimport re\nfrom datetime import date, timedelta\nfrom urllib.parse import urlencode\nimport asyncio, aiohttp\nfrom tqdm.asyncio import tqdm_asyncio\nfrom tqdm import tqdm\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:29:00.220893Z","iopub.execute_input":"2026-01-17T01:29:00.221235Z","iopub.status.idle":"2026-01-17T01:29:00.226830Z","shell.execute_reply.started":"2026-01-17T01:29:00.221208Z","shell.execute_reply":"2026-01-17T01:29:00.225822Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# input","metadata":{}},{"cell_type":"code","source":"BASE_URL = \"https://www.boatrace.jp\"\n\ndef parse_weather(soup):\n    weather = []\n    for unit in soup.select(\"div.weather1_bodyUnit\"):\n        title = unit.select_one(\".weather1_bodyUnitLabelTitle\")\n        data  = unit.select_one(\".weather1_bodyUnitLabelData\")\n        if title:\n            weather.append(title.text.strip() if data is None else f\"{title.text.strip()}:{data.text.strip()}\")\n        \n    wind_dir = None\n    for p in soup.select(\"p.weather1_bodyUnitImage\"):\n        for c in p.get(\"class\", []):\n            m = re.match(r\"is-wind(\\d+)\", c)\n            if m:\n                wind_dir = int(m.group(1))\n\n    weather.append(f\"風向:{wind_dir}\" if wind_dir is not None else \"風向:\")\n    return weather\n\ndef parse_order(soup):\n    order = []\n    for div in soup.select(\"div.table1_boatImage1\"):\n        s = div.select_one(\"span.table1_boatImage1Number\")\n        if s:\n            n = int(s.text.strip())\n            if n not in order:\n                order.append(n)\n    return order\n\ndef to_num(s):\n    x = re.sub(r\"[^0-9.]\", \"\", s)\n    return float(x) if x else None\n\nasync def fetch(session, path):\n    try:\n        async with session.get(path) as r:\n            r.raise_for_status()\n            html = await r.text()\n\n        soup = BeautifulSoup(html, \"lxml\")\n        # return soup\n        # tr/td\n        ret = [\n            [td.get_text(\"\\n\", strip=True) for td in tr.select(\"td\") if td.get_text(strip=True)]\n            for tr in soup.select(\"tr\")\n        ]\n\n        if \"beforeinfo\" in path:\n            order = parse_order(soup)\n            ret.append(order)\n            \n            w = parse_weather(soup)\n            # \"気温:22.0℃\" -> 22.0, \"晴\" -> そのまま, \"風向:3\" -> 3\n            w2 = []\n            for s in w:\n                if \":\" in s:\n                    k, v = s.split(\":\", 1)\n                    # 天気(晴/曇)は数値にできないので文字で残す\n                    vv = to_num(v) if to_num(v) is not None else v.strip() or None\n                    w2.append(vv)\n                else:\n                    w2.append(s)\n            ret.append(w2)\n        return ret\n\n    except Exception as e:\n        return e\n\nasync def main(paths, conn_limit=100, timeout_sec=30, chunk_size=200):\n    timeout = aiohttp.ClientTimeout(total=timeout_sec)\n    connector = aiohttp.TCPConnector(limit=conn_limit)\n\n    results_all = []\n\n    async with aiohttp.ClientSession(\n        base_url=BASE_URL,\n        timeout=timeout,\n        connector=connector,\n        headers={\"User-Agent\": \"Mozilla/5.0\"}\n    ) as session:\n\n        for i in tqdm(range(0, len(paths), chunk_size)):\n            chunk = paths[i:i+chunk_size]\n\n            results = await asyncio.gather(\n                *(fetch(session, p) for p in chunk)\n            )\n\n            results_all.extend(results)\n\n    return results_all","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:29:01.795475Z","iopub.execute_input":"2026-01-17T01:29:01.795818Z","iopub.status.idle":"2026-01-17T01:29:01.811357Z","shell.execute_reply.started":"2026-01-17T01:29:01.795791Z","shell.execute_reply":"2026-01-17T01:29:01.810064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def to_int(x):\n    try:\n        return int(x)\n    except:\n        return 0\n\ndef to_float(x):\n    try:\n        return float(x)\n    except:\n        return 0.0\n\ndef find_int(pattern, text):\n    m = re.search(pattern, text)\n    return int(m.group(1)) if m else 0\n\ndef find_float(pattern, text):\n    m = re.search(pattern, text)\n    return float(m.group(1)) if m else 0.0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:26:17.473050Z","iopub.execute_input":"2026-01-17T01:26:17.473663Z","iopub.status.idle":"2026-01-17T01:26:17.918463Z","shell.execute_reply.started":"2026-01-17T01:26:17.473627Z","shell.execute_reply":"2026-01-17T01:26:17.917507Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# get_racelist","metadata":{}},{"cell_type":"code","source":"def get_racelist(texts):\n    texts = [text[1:7] for text in texts if len(text)>1 and len(text[1])>=10]\n    res = {}\n    rank = {'A1': 4, 'A2': 3, 'B1': 2, 'B2': 1}\n\n    for i, row in enumerate(texts):\n        s = row[0]\n        s = list(map(str, s.split('\\n')))\n        player_id = s[0]\n        grade = rank.get(s[2], 0)\n        age = to_int(s[5].split('/')[0][:-1])\n        weight = to_float(s[5].split('/')[1][:-2])\n\n        f_l = row[1]\n        f = find_int(r\"F(\\d+)\", f_l)\n        l = find_int(r\"L(\\d+)\", f_l)\n        st = find_float(r\"(\\d+\\.\\d+)\", f_l)\n        \n        avg1, rate1, rate2 = map(to_float, row[2].split('\\n'))\n        avg2, rate3, rate4 = map(to_float, row[3].split('\\n'))\n        _, mot1, mot2 = map(to_float, row[4].split('\\n'))\n        _, bot1, bot2 = map(to_float, row[5].split('\\n'))\n\n\n    # return [grade,age,weight,f,l,st,avg1,rate1,\n    #         rate2,avg2,rate3,rate4,mot1,mot2,bot1,bot2]\n    \n        res[i+1] = {\n            \"player_id\": player_id,\n            \"grade\": grade,\n            \"age\": age,\n            \"weight\": weight,\n            \"f\": f,\n            \"l\": l,\n            \"st\": st,\n            \"avg1\": avg1,\n            \"rate1\": rate1,\n            \"rate2\": rate2,\n            \"avg2\": avg2,\n            \"rate3\": rate3,\n            \"rate4\": rate4,\n            \"mot1\": mot1,\n            \"mot2\": mot2,\n            \"bot1\": bot1,\n            \"bot2\": bot2,\n        }\n    return res\n\ndef odds3t(texts):\n    ODDS = {}\n    texts = [t for text in texts for t in text]\n    texts = map(str, texts)\n\n    for _ in range(5):\n        A = []\n        for i in range(1, 7):\n            a = to_int(next(texts, 0))\n            b = to_int(next(texts, 0))\n            c = to_float(next(texts, 0))\n            ODDS[(1, i, a, b)] = c\n            A.append(a)\n\n        for _ in range(3):\n            for i in range(1, 7):\n                b = to_int(next(texts, 0))\n                c = to_float(next(texts, 0))\n                ODDS[(1, i, A[i-1], b)] = c\n    return ODDS\n\n\ndef odds3f(texts):\n    ODDS = {}\n    for text in texts:\n        text = list(map(str, text))\n        if len(text) == 3 or (len(text) > 3 and text[0] == text[3]):\n            a = to_int(text[0])\n            text = [text[i] for i in range(len(text)) if i % 3 != 0]\n\n        for b in range(1, len(text)//2 + 1):\n            c = to_int(text[2*(b-1)])\n            d = to_float(text[2*(b-1)+1])\n            ODDS[(2, b, a, c)] = d\n    return ODDS\n\n\ndef odds2tf(texts):\n    ODDS = {}\n    for text in texts[:5]:\n        text = list(map(str, text))\n        for a in range(6):\n            b = to_int(text[2*a]) if 2*a < len(text) else 0\n            c = to_float(text[2*a+1]) if 2*a+1 < len(text) else 0.0\n            ODDS[(3, a+1, b)] = c\n\n    for text in texts[6:]:\n        text = list(map(str, text))\n        for a in range(len(text)//2):\n            b = to_int(text[2*a])\n            c = to_float(text[2*a+1])\n            ODDS[(4, a+1, b)] = c\n    return ODDS\n\n\ndef oddsk(texts):\n    ODDS = {}\n    for text in texts[:5]:\n        text = list(map(str, text))\n        for a in range(len(text)//2):\n            b = to_int(text[2*a])\n            c0 = text[2*a+1]\n            m = re.findall(r\"[\\d.]+\", c0)\n            if len(m) == 2:\n                c = to_float(m[0])\n            else:\n                c = to_float(c0)\n            ODDS[(5, a+1, b)] = c\n    return ODDS\n\ndef oddstf(texts):\n    ODDS = {}\n    for text in texts[:6]:\n        a = to_int(text[0])\n        b = str(text[2])\n        ODDS[(6, a)] = to_float(b)\n\n    for text in texts[7:]:\n        a = to_int(text[0])\n        b = str(text[2])\n        m = re.findall(r\"[\\d.]+\", b)\n        ODDS[(7, a)] = to_float(m[0]) if m else 0.0\n    return ODDS\n\n\ndef odds_total(TEXTS):\n    ODDS = {}\n    for texts, f in zip(TEXTS, [odds3t, odds3f, odds2tf, oddsk, oddstf]):\n        res = f(texts[3:])\n        if res is False:\n            return False\n        ODDS |= res\n    return ODDS\n\ndef get_beforeinfo(texts):\n    ret = [0.0]*6\n    for i in range(6):\n        if len(texts[4+i*4])>3 and to_float(texts[4+i*4][3])!=0.0:\n            ret[i] = to_float(texts[4+i*4][3])\n\n    return [ret[i] if ret[i]!=0.0 else 10.0 for i in range(6)] + texts[-1], texts[-2]\n\ndef get_result(rows):\n    res = []\n    tof = True\n    for row in rows:\n        if not row:\n            continue\n        if row[0]==\"3連単\":\n            return tuple(map(int, re.findall(r\"\\d+\", row[1])))\n\ndef make_date(st=0, days=5):\n    end = date(2026, 1, 3)\n    end = end - timedelta(days=st)\n    start = end - timedelta(days=days)\n    \n    dates = []\n    \n    d = start\n    while d < end:\n        for jcd in range(1, 25):\n            for rno in range(1, 7):\n                params = {\n                    \"rno\": rno,\n                    \"jcd\": f\"{jcd:02d}\",\n                    \"hd\": d.strftime(\"%Y%m%d\"),\n                }\n                dates.append(urlencode(params))\n        d += timedelta(days=1)\n        \n    return dates\n\ndef make_url(kind, date):\n    return f\"https://www.boatrace.jp/owpc/pc/race/{kind}?{date}\"\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:26:17.919855Z","iopub.execute_input":"2026-01-17T01:26:17.920130Z","iopub.status.idle":"2026-01-17T01:26:17.947856Z","shell.execute_reply.started":"2026-01-17T01:26:17.920106Z","shell.execute_reply":"2026-01-17T01:26:17.946821Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n\nt0 = time.perf_counter()\n\nleft, right = 216, 217\ndates = make_date(left, right)\n# dates = ['rno=3&jcd=07&hd=20250531']\n# dates = dates[:50]\n\npaths = [make_url('racelist', D) for D in dates]\nRACELIST = await main(paths)\nprint('racelist')\n\nodds_names = ['odds3t', 'odds3f', 'odds2tf', 'oddsk', \"oddstf\"]\npaths = [make_url(name, D) for D in dates for name in odds_names]\nODDS = await main(paths)\nODDS = [ODDS[i*5:i*5+5] for i in range(len(dates))]\nprint('odds')\n\npaths = [make_url('raceresult', D) for D in dates]\nRACERESULT = await main(paths)\nprint('raceresult')\n\npaths = [make_url('beforeinfo', D) for D in dates]\nBEFOREINFO = await main(paths)\nprint('beforeinfo')\n\nt1 = time.perf_counter()\nprint(f\"{t1 - t0:.3f} sec\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(RACELIST), len(ODDS), len(RACERESULT), len(BEFOREINFO), len(dates))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:27:57.999141Z","iopub.execute_input":"2026-01-17T01:27:57.999829Z","iopub.status.idle":"2026-01-17T01:27:58.005276Z","shell.execute_reply.started":"2026-01-17T01:27:57.999802Z","shell.execute_reply":"2026-01-17T01:27:58.004158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = {}\ncnt = 0\nfor racelist, odds, raceresult, beforeinfo, D in zip(RACELIST, ODDS, RACERESULT, BEFOREINFO, dates):\n    # print(1)\n    if not racelist: continue\n        \n    data = {}\n    try:\n        data['racelist'] = get_racelist(racelist[3:])\n    except:\n        print(1, D)\n        cnt += 1\n        continue\n\n    try:\n        data['beforeinfo'], data['order'] = get_beforeinfo(beforeinfo)\n        m = re.search(r\"rno=(\\d{1,2})\", D)\n        rno = int(m.group(1))\n        data['time'] = beforeinfo[1][rno]\n    except:\n        print(2, D)\n        cnt += 1\n        continue\n        \n    \n    try:\n        data['odds'] = odds_total(odds)\n    except:\n        print(3, D)\n        cnt += 1\n        continue\n    \n    try:\n        data['raceresult'] = get_result(raceresult[3:])\n    except:\n        print(4, D)\n        cnt += 1\n        continue\n\n    X[D] = data\n\nprint(cnt)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:27:58.006480Z","iopub.execute_input":"2026-01-17T01:27:58.006885Z","iopub.status.idle":"2026-01-17T01:27:58.027194Z","shell.execute_reply.started":"2026-01-17T01:27:58.006859Z","shell.execute_reply":"2026-01-17T01:27:58.026260Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\npickle.dump(X, open(\"boatrace_input.pkl\", \"wb\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-17T01:26:17.981767Z","iopub.status.idle":"2026-01-17T01:26:17.982112Z","shell.execute_reply.started":"2026-01-17T01:26:17.981943Z","shell.execute_reply":"2026-01-17T01:26:17.981968Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}