From eb9679e1d41ba02a572c9a6bda476cc49074c13a Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Thu, 1 Oct 2026 02:04:27 +0900 Subject: [PATCH 1/4] =?UTF-8?q?=E5=88=B0=E7=9D=80=E6=99=82=E9=96=93?= =?UTF-8?q?=E6=8E=A8=E5=AE=9A=E3=82=92=E5=AE=9F=E9=9A=9B=E3=81=AE=E6=89=80?= =?UTF-8?q?=E8=A6=81=E6=99=82=E9=96=93=E3=81=A8=E6=AF=94=E3=81=B9=E3=82=8B?= =?UTF-8?q?=E3=83=99=E3=83=B3=E3=83=81=E3=83=9E=E3=83=BC=E3=82=AF=E3=82=92?= =?UTF-8?q?=E8=BF=BD=E5=8A=A0=E3=81=97=E3=80=81CI=E3=81=A7=E6=8E=A8?= =?UTF-8?q?=E5=AE=9A=E3=81=AE=E6=82=AA=E5=8C=96=E3=82=92=E6=A4=9C=E5=87=BA?= =?UTF-8?q?=E3=81=A7=E3=81=8D=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB=E3=81=97?= =?UTF-8?q?=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 1 + Makefile | 10 +- README.md | 5 + docs/architecture.md | 15 ++ scripts/travel_time_report.py | 107 ++++++++++++++ src/lib.rs | 2 + src/travel_times.rs | 262 ++++++++++++++++++++++++++++++++++ travel_times/README.md | 66 +++++++++ travel_times/baseline.csv | 14 ++ travel_times/cases.csv | 19 +++ 10 files changed, 500 insertions(+), 1 deletion(-) create mode 100644 scripts/travel_time_report.py create mode 100644 src/travel_times.rs create mode 100644 travel_times/README.md create mode 100644 travel_times/baseline.csv create mode 100644 travel_times/cases.csv diff --git a/AGENTS.md b/AGENTS.md index 0a9ded7b..9dd2688f 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -64,6 +64,7 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - **Schema** – Changing a GraphQL type changes the SDL. Update `schema/public.graphql` in the same change; CI compares it against the running Worker's `/__schema` and fails on any difference. That diff is exactly the client-visible impact. - **Data verification** – Execute `cargo run -p data_validator` whenever CSVs change and record results in pull requests. - **IPA coverage audit** – Execute `make ipa-audit` when English or romanized CSV names change. This is a read-only report for `data/2!lines.csv`, `data/3!stations.csv`, and `data/4!types.csv`; it does not fail validation, but highlights unresolved tokens and example names so the IPA dictionary can be extended deliberately. +- **Travel-time benchmark** – `travel_times/cases.csv` lists real travel times (a range in minutes, weekday daytime, trains with the same stopping pattern as the line group) that the arrival estimation is measured against. `cargo test -p stationapi-worker` (`src/travel_times.rs`) fails when any case moves more than 1 percentage point further outside its range than `travel_times/baseline.csv` records, or when the mean gets worse; CI runs on `data/*.csv`, so cases whose line group exists only in `generated/` are skipped. `make travel-time-report` (`TRAVEL_TIME_API`, default the local `make dev` Worker) measures every case against a Worker built from the generated data. A change to the speed tables, the general speed rules, or the estimation parameters must attach the report from before and after, and must not trade one line's accuracy for the whole set's; after an intended change, regenerate the baseline with `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times`. Values come from open GTFS feeds (credit them in the README's Data Sources) or from the maintainer. - **Endpoint benchmarks** – `make bench` (or `python3 .claude/skills/benchmark-gql/bench.py`) replays every `Query` field against production (`gql.trainlcd.app`, script `stationapi`) and staging (`gql-stg.trainlcd.app`, script `stationapi-stg`) and writes a Markdown report under `benchmarks/`. Both environments embed the same data, so any difference is implementation — which makes this the way to see what a `dev`-to-`master` release will do to performance before it ships. Besides client latency it records the Worker's `cpuTime`, read from `wrangler tail --format json` and matched to each request by `cf-ray`; the tail is filtered on a per-run request header, so production's live traffic does not leak into the sample. Collecting CPU time needs the `workers_tail (read)` scope, and the run sends hundreds of real requests to production — it is not a routine check. Add a case to `.claude/skills/benchmark-gql/queries.json` whenever a `Query` field is added, and never edit an existing case's variables: the reports are meant to stay comparable across runs. ## GraphQL Query Overview diff --git a/Makefile b/Makefile index 779fd12b..da433b6c 100644 --- a/Makefile +++ b/Makefile @@ -1,7 +1,7 @@ # StationAPI Makefile # よく使うタスクの定義 -.PHONY: help test check fmt clippy data build dev deploy deploy-production schema ipa-audit bench clean +.PHONY: help test check fmt clippy data build dev deploy deploy-production schema ipa-audit bench travel-time-report clean # CI (.github/workflows/build_worker.yml) と同じ版を使う。グローバルへ入れて # いなくても npx が取ってくるので、版ずれでビルド結果が変わらない。 @@ -22,6 +22,7 @@ help: @echo " schema - Diff the running Worker's SDL against schema/public.graphql" @echo " ipa-audit - Print IPA coverage report for English/romanized CSV names" @echo " bench - Compare production vs staging GraphQL performance (sends live traffic to both)" + @echo " travel-time-report - Compare estimated travel times with travel_times/cases.csv (TRAVEL_TIME_API, default http://127.0.0.1:8787/)" @echo " clean - Clean build artifacts" @echo "" @echo "Environment variables:" @@ -84,6 +85,13 @@ ipa-audit: # 実在のエンドポイントへ数百リクエスト投げるので、気軽に回すものではない。 # CPU Time の収集には wrangler の workers_tail (read) 権限が要る。 # 追加の引数は BENCH_ARGS で渡す (例: make bench BENCH_ARGS="--repeat 30")。 +# 到着時間推定の所要時間を、実際の所要時間 (travel_times/cases.csv) と比べる。 +# 本番と同じ生成データで測るため、`make data && make dev` で起動した Worker か +# ステージングに向ける (TRAVEL_TIME_API)。 +TRAVEL_TIME_API ?= http://127.0.0.1:8787/ +travel-time-report: + python3 scripts/travel_time_report.py --api $(TRAVEL_TIME_API) + bench: @echo "警告: 本番 (gql.trainlcd.app) とステージングへ実リクエストを送ります。" >&2 @echo " 既定で 1 環境あたり 400 件超、うち数十件は Worker の CPU を 500ms 以上使います。" >&2 diff --git a/README.md b/README.md index 12a26397..fc3a17de 100644 --- a/README.md +++ b/README.md @@ -31,6 +31,11 @@ This project includes a comprehensive dataset of Japanese railway information in (https://nlftp.mlit.go.jp/ksj/gml/datalist/KsjTmplt-N02-2025.html) を加工して作成 - Bus stops and routes are derived from the GTFS and ODPT feeds listed in `preprocessor/src/gtfs/feed.rs` and `preprocessor/src/gtfs/odpt.rs`. +- Some of the reference travel times in `travel_times/cases.csv` are derived from + the Toei Subway GTFS (Bureau of Transportation, Tokyo Metropolitan Government, + CC BY 4.0) and from the Tokyo Metro and Metropolitan Intercity Railway + (Tsukuba Express) GTFS feeds published by the Public Transportation Open Data + Center under the Public Transportation Open Data Basic License. ## Contributors ✨ diff --git a/docs/architecture.md b/docs/architecture.md index 3629a332..3d1eb55b 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -506,6 +506,21 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int! - 両端の駅が `fromStationId` / `toStationId` と一致しない - `viaLineIds`・`directionId`・`lineGroupId` と同時に指定されている +### 所要時間のベンチマーク (`travel_times/`) + +到着時間推定の所要時間を、実際の列車の所要時間と比べる基準を `travel_times/cases.csv` +に置いています。速度の較正テーブルや一般則は、1 つの路線に合わせて変えると、同じ +規則を使うほかの路線の推定も変わります。変更の前後で全体の誤差を測るための仕組み +です。 + +- `cargo test -p stationapi-worker` (`src/travel_times.rs`) は、基準ごとの「実際の範囲 + からの外れ」を `travel_times/baseline.csv` の記録と比べ、悪くなると失敗します。 + CI は `data/*.csv` で動くので、生成データにしか無い種別グループの基準は飛ばします。 +- `make travel-time-report` は、生成データで動く Worker に問い合わせて全件の誤差を + 出します。推定の規則や較正を変える PR には、変更前と変更後のレポートを載せます。 + +基準の決め方と記録の更新方法は `travel_times/README.md` にあります。 + ### 行き先の検索 (`stationsByName`) `stationsByName` に `fromStationGroupId` を指定すると、その駅から行ける駅だけに diff --git a/scripts/travel_time_report.py b/scripts/travel_time_report.py new file mode 100644 index 00000000..d4d0d049 --- /dev/null +++ b/scripts/travel_time_report.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +"""実際の所要時間 (travel_times/cases.csv) に対する到着時間推定の誤差を、動いている +Worker に問い合わせて Markdown で出す。 + +CI の回帰テスト (src/travel_times.rs) は data/*.csv だけで動くので、生成データにしか +無い種別グループを飛ばし、線路の長さも持たない。本番と同じ生成データでの精度は、 +`make data && make dev` で起動した Worker か、ステージングに向けてこれで測る。 +推定の規則や較正を変える PR には、変更前と変更後のこのレポートを載せる。 + +使い方: + python3 scripts/travel_time_report.py # http://127.0.0.1:8787/ + python3 scripts/travel_time_report.py --api +""" +from __future__ import annotations + +import argparse +import csv +import json +import statistics +import sys +import urllib.request +from pathlib import Path + +CASES = Path(__file__).resolve().parent.parent / "travel_times" / "cases.csv" +# 既定の Python-urllib は配信側で弾かれるので、bench.py と同じく名乗る +USER_AGENT = "stationapi-travel-time-report/1.0 (+https://github.com/TrainLCD/StationAPI)" +QUERY = """query TravelTimeReport($from: Int!, $to: Int!, $group: Int!) { + estimateArrivalTimes(fromStationId: $from, toStationId: $to, + legs: [{ lineGroupId: $group, fromStationId: $from, toStationId: $to }]) { + routes { stops { stationId cumulativeMinutes departureCumulativeMinutes } } + } +}""" + + +def estimate(api: str, case: dict) -> float: + end = int(case["slice_end_station_id"] or case["to_station_id"]) + body = json.dumps({ + "query": QUERY, + "variables": { + "from": int(case["from_station_id"]), + "to": end, + "group": int(case["line_group_id"]), + }, + }).encode() + req = urllib.request.Request( + api, data=body, headers={"content-type": "application/json", "user-agent": USER_AGENT} + ) + with urllib.request.urlopen(req, timeout=30) as res: + data = json.load(res) + if data.get("errors"): + raise RuntimeError("; ".join(e.get("message", "") for e in data["errors"])) + stops = data["data"]["estimateArrivalTimes"]["routes"][0]["stops"] + target = int(case["to_station_id"]) + stop = next(s for s in stops[1:] if s["stationId"] == target) + key = "cumulativeMinutes" if case["measure"] == "arrival" else "departureCumulativeMinutes" + return float(stop[key]) + + +def range_error(est: float, lo: float, hi: float) -> float: + if est < lo: + return (lo - est) / lo + if est > hi: + return (est - hi) / hi + return 0.0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n")[0]) + parser.add_argument("--api", default="http://127.0.0.1:8787/") + args = parser.parse_args() + + with CASES.open(encoding="utf-8") as f: + cases = list(csv.DictReader(f)) + + rows, range_errors, mid_errors, failed = [], [], [], [] + for case in cases: + lo, hi = float(case["real_min_minutes"]), float(case["real_max_minutes"]) + try: + est = estimate(args.api, case) + except Exception as e: # noqa: BLE001 - 1 件の失敗で全体を止めない + failed.append(f"{case['label']}: {e}") + continue + mid = (lo + hi) / 2 + r, m = range_error(est, lo, hi), (est - mid) / mid + range_errors.append(r) + mid_errors.append(abs(m)) + rows.append( + f"| {case['label']} | {lo:g}〜{hi:g}分 | {est:.1f}分 | {r * 100:.1f}% | {m * 100:+.1f}% |" + ) + + print(f"# 到着時間推定の誤差 ({args.api})\n") + print("| 基準 | 実際 | 推定 | 範囲からの外れ | 範囲の中央からのずれ |") + print("| --- | --- | --- | --- | --- |") + print("\n".join(rows)) + if range_errors: + print( + f"\n{len(range_errors)} 件: 範囲からの外れの平均 {statistics.mean(range_errors) * 100:.2f}%、" + f"範囲の中央からのずれ (絶対値) の平均 {statistics.mean(mid_errors) * 100:.2f}%" + ) + if failed: + print("\n推定できなかった基準:\n") + print("\n".join(f"- {line}" for line in failed)) + return 0 if range_errors else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/lib.rs b/src/lib.rs index f7f850f4..f7a140c7 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -13,6 +13,8 @@ mod graphql; mod index; mod repository; +#[cfg(test)] +mod travel_times; use async_graphql::http::GraphiQLSource; use async_graphql::Request as GqlRequest; diff --git a/src/travel_times.rs b/src/travel_times.rs new file mode 100644 index 00000000..0ddf4945 --- /dev/null +++ b/src/travel_times.rs @@ -0,0 +1,262 @@ +//! 実際の所要時間 (`travel_times/cases.csv`) に対する到着時間推定の回帰の見張り。 +//! +//! 基準ごとに推定の所要時間を出し、実際の範囲からの外れ (範囲内なら 0、外れたら +//! 近い端からの割合) を求める。記録した推定 (`travel_times/baseline.csv`) より +//! 悪くなった基準があるか、平均が悪くなったら失敗にする。速度の較正や一般則を +//! 1 つの路線に合わせて変えたときに、ほかの路線がどれだけ崩れたかをここで見る。 +//! +//! CI は `generated/` を作らずに `data/*.csv` で動くので、生成データにしか無い +//! 種別グループ (各駅停車を補う系統など) の基準は飛ばす。本番と同じデータでの +//! 精度は `make travel-time-report` で測る。 +//! +//! 推定を意図して変えたときは、次で記録を更新する。 +//! `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times` + +use std::collections::HashMap; + +use stationapi::domain::repository::station_repository::StationRepository; +use stationapi::model::RouteLegRequest; +use stationapi::use_case::traits::query::QueryUseCase; + +use crate::repository::MemStationRepository; + +const CASES: &str = include_str!("../travel_times/cases.csv"); +const BASELINE_PATH: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/travel_times/baseline.csv"); +/// 1 基準あたり、範囲からの外れがこれだけ増えたら失敗にする (割合) +const CASE_TOLERANCE: f64 = 0.01; +/// 平均の比較の許容幅 (割合)。記録は推定の分を小数 4 桁に丸めて書くので、その丸めで +/// 平均がわずかに動くぶんを吸収する +const MEAN_TOLERANCE: f64 = 1e-4; + +#[derive(Debug, Clone, Copy, PartialEq)] +enum Measure { + /// 出発駅の発車から到着駅の到着まで + Arrival, + /// 出発駅の発車から到着駅の発車まで (到着時刻を載せない時刻表向け) + Departure, +} + +#[derive(Debug)] +struct Case { + label: String, + line_group_id: u32, + from_station_id: u32, + to_station_id: u32, + /// 推定する区間の終わり。`Departure` では到着駅の先まで推定しないと + /// 到着駅が終点になり、停車時間が付かない + slice_end_station_id: u32, + measure: Measure, + real_min: f64, + real_max: f64, +} + +impl Case { + /// 実際の範囲からの外れ。範囲内なら 0、外れたら近い端に対する割合 + fn range_error(&self, estimated: f64) -> f64 { + if estimated < self.real_min { + (self.real_min - estimated) / self.real_min + } else if estimated > self.real_max { + (estimated - self.real_max) / self.real_max + } else { + 0.0 + } + } +} + +fn parse_cases() -> Vec { + let mut lines = CASES.lines(); + let header: Vec<&str> = lines.next().expect("見出しの行が無い").split(',').collect(); + let col = |name: &str| { + header + .iter() + .position(|h| *h == name) + .unwrap_or_else(|| panic!("列 {name} が無い")) + }; + let (label, group, from, to, end, measure, min, max) = ( + col("label"), + col("line_group_id"), + col("from_station_id"), + col("to_station_id"), + col("slice_end_station_id"), + col("measure"), + col("real_min_minutes"), + col("real_max_minutes"), + ); + lines + .filter(|line| !line.trim().is_empty()) + .map(|line| { + let f: Vec<&str> = line.split(',').collect(); + assert_eq!(f.len(), header.len(), "列の数が見出しと違う: {line}"); + let num = |i: usize| -> u32 { + f[i].parse() + .unwrap_or_else(|_| panic!("数値ではない: {line}")) + }; + let minutes = |i: usize| -> f64 { + f[i].parse() + .unwrap_or_else(|_| panic!("数値ではない: {line}")) + }; + let measure = match f[measure] { + "arrival" => Measure::Arrival, + "departure" => Measure::Departure, + other => panic!("measure は arrival / departure のどちらか: {other}"), + }; + let to_station_id = num(to); + Case { + label: f[label].to_string(), + line_group_id: num(group), + from_station_id: num(from), + to_station_id, + slice_end_station_id: if f[end].is_empty() { + to_station_id + } else { + num(end) + }, + measure, + real_min: minutes(min), + real_max: minutes(max), + } + }) + .collect() +} + +/// repository の実装は await しない (索引を引くだけ) ので、1 回 poll すれば終わる +fn block_on(future: F) -> F::Output { + let mut context = std::task::Context::from_waker(std::task::Waker::noop()); + match std::pin::pin!(future).poll(&mut context) { + std::task::Poll::Ready(value) => value, + std::task::Poll::Pending => panic!("repository futures complete without waiting"), + } +} + +/// 推定の所要時間 (分)。種別グループがこのデータに無ければ `None` +fn estimate(case: &Case) -> Option { + let group = block_on(MemStationRepository.get_by_line_group_id(case.line_group_id)).ok()?; + if group.is_empty() { + return None; + } + let legs = [RouteLegRequest { + line_group_id: case.line_group_id, + from_station_id: case.from_station_id, + to_station_id: case.slice_end_station_id, + }]; + let stops = block_on(crate::interactor().estimate_connected_route_arrival_times(&legs)) + .unwrap_or_else(|e| panic!("{}: 推定できない: {e}", case.label)); + let stop = stops + .iter() + .skip(1) + .find(|stop| stop.station_cd as u32 == case.to_station_id) + .unwrap_or_else(|| panic!("{}: 推定の駅列に到着駅が無い", case.label)); + Some(match case.measure { + Measure::Arrival => stop.cumulative_minutes, + Measure::Departure => stop.departure_cumulative_minutes, + }) +} + +fn read_baseline() -> HashMap { + let text = std::fs::read_to_string(BASELINE_PATH).unwrap_or_default(); + text.lines() + .skip(1) + .filter(|line| !line.trim().is_empty()) + .map(|line| { + let (label, minutes) = line.rsplit_once(',').expect("label,estimated_minutes の形"); + ( + label.to_string(), + minutes.parse().expect("推定の分が数値ではない"), + ) + }) + .collect() +} + +#[test] +fn cases_are_well_formed() { + let cases = parse_cases(); + assert!(!cases.is_empty()); + let mut labels = std::collections::HashSet::new(); + for case in &cases { + assert!(labels.insert(&case.label), "label が重複: {}", case.label); + assert!( + case.real_min > 0.0 && case.real_min <= case.real_max, + "{}", + case.label + ); + if case.measure == Measure::Departure { + assert_ne!( + case.slice_end_station_id, case.to_station_id, + "{}: departure は到着駅の先まで推定する (slice_end_station_id)", + case.label + ); + } + } +} + +#[test] +fn estimates_do_not_drift_away_from_real_travel_times() { + let cases = parse_cases(); + let estimated: Vec<(&Case, Option)> = cases.iter().map(|c| (c, estimate(c))).collect(); + + let mut report = vec![String::from("\n基準 | 実際 | 推定 | 範囲からの外れ")]; + for (case, est) in &estimated { + let real = format!("{}〜{}分", case.real_min, case.real_max); + report.push(match est { + Some(est) => format!( + "{} | {real} | {est:.1}分 | {:.1}%", + case.label, + case.range_error(*est) * 100.0 + ), + None => format!( + "{} | {real} | (このデータに種別グループが無い) | -", + case.label + ), + }); + } + println!("{}", report.join("\n")); + + if std::env::var("TRAVEL_TIMES_UPDATE_BASELINE").as_deref() == Ok("1") { + let mut out = String::from("label,estimated_minutes\n"); + for (case, est) in &estimated { + if let Some(est) = est { + out.push_str(&format!("{},{est:.4}\n", case.label)); + } + } + std::fs::write(BASELINE_PATH, out).expect("記録を書き込めない"); + return; + } + + let baseline = read_baseline(); + let (mut now_sum, mut base_sum, mut n) = (0.0, 0.0, 0); + let mut worse = Vec::new(); + for (case, est) in &estimated { + let Some(est) = est else { continue }; + let base = baseline.get(&case.label).unwrap_or_else(|| { + panic!( + "{}: 記録が無い。TRAVEL_TIMES_UPDATE_BASELINE=1 で記録を更新する", + case.label + ) + }); + let (now_err, base_err) = (case.range_error(*est), case.range_error(*base)); + if now_err > base_err + CASE_TOLERANCE { + worse.push(format!( + "{}: 範囲からの外れが {:.1}% → {:.1}% (推定 {base:.1}分 → {est:.1}分)", + case.label, + base_err * 100.0, + now_err * 100.0 + )); + } + now_sum += now_err; + base_sum += base_err; + n += 1; + } + assert!(n > 0, "このデータで推定できる基準が 1 つも無い"); + assert!( + worse.is_empty(), + "実際の所要時間から離れた基準がある:\n{}", + worse.join("\n") + ); + let (now_mean, base_mean) = (now_sum / n as f64, base_sum / n as f64); + assert!( + now_mean <= base_mean + MEAN_TOLERANCE, + "範囲からの外れの平均が {:.2}% → {:.2}% に悪化した", + base_mean * 100.0, + now_mean * 100.0 + ); +} diff --git a/travel_times/README.md b/travel_times/README.md new file mode 100644 index 00000000..7d81d0fb --- /dev/null +++ b/travel_times/README.md @@ -0,0 +1,66 @@ +# travel_times/ + +到着時間推定 (`stationapi/src/domain/arrival_estimation.rs`) の所要時間を、実際の +列車の所要時間と比べるための基準を置く場所です。速度の較正テーブルや一般則は、 +1 つの路線に合わせて変えると、同じ規則を使うほかの路線の推定も変わります。 +変更の前後で全体の誤差を測り、局所的な合わせ込みで全体が崩れないようにします。 + +## ファイル + +| パス | 内容 | +| --- | --- | +| `cases.csv` | 基準の一覧。1 行が 1 区間 | +| `baseline.csv` | CI のデータ (`data/*.csv`) で出した推定の所要時間の記録 | + +`cases.csv` の列は次のとおりです。 + +| 列 | 内容 | +| --- | --- | +| `label` | 基準の名前。一覧の中で重複させない | +| `line_group_id` | 推定に使う種別グループ (系統) | +| `from_station_id` / `to_station_id` | 区間の出発駅と到着駅 | +| `slice_end_station_id` | 推定する区間の終わり。`measure` が `departure` のときに、到着駅より先の駅を書く。空なら到着駅 | +| `measure` | `arrival` は出発駅の発車から到着駅の到着まで。`departure` は到着駅の発車までで、到着時刻を載せない時刻表の値に使う | +| `real_min_minutes` / `real_max_minutes` | 実際の所要時間の範囲 (分) | +| `source` | 値の出どころ | + +## 基準の決め方 + +- 平日の日中に出発駅を出る列車の所要時間を使います。途中で待ち合わせる列車などで + 所要時間に幅があるときは、最小と最大をそのまま書きます。 +- 列車は、`line_group_id` の種別グループと同じ停車パターンのものに限ります。 + 停車駅が違う列車の所要時間と比べると、推定の誤差ではない差が混ざります。 +- 出どころは、公開 GTFS か、メンテナが確認した値に限ります。GTFS から求めた値を + 足すときは、そのフィードの出典をリポジトリ直下の README の「Data Sources」に + 書きます。 + +## 測る + +### CI (回帰の見張り) + +`cargo test -p stationapi-worker` の `travel_times` モジュール (`src/travel_times.rs`) +が、基準ごとに推定を出し、実際の範囲からの外れを求めます。範囲に入っていれば 0、 +外れていれば近い端に対する割合です。`baseline.csv` の記録と比べて、1 件でも +外れが 1 ポイントより多く増えるか、平均が悪くなると失敗します。 + +CI は `generated/` を作らずに `data/*.csv` で動くので、生成データにしか無い種別 +グループ (各駅停車を補う系統など) の基準は飛ばします。 + +推定を意図して変えたときは、記録を更新して同じ PR に含めます。 + +```bash +TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times +``` + +### レポート (本番と同じデータでの精度) + +動いている Worker に `estimateArrivalTimes` を問い合わせ、全件の誤差を Markdown で +出します。`make data && make dev` で起動した Worker (既定) か、ステージングに +向けます。 + +```bash +make travel-time-report +make travel-time-report TRAVEL_TIME_API=https://gql-stg.trainlcd.app/ +``` + +推定の規則や較正を変える PR には、変更前と変更後のレポートを載せます。 diff --git a/travel_times/baseline.csv b/travel_times/baseline.csv new file mode 100644 index 00000000..de4a0b93 --- /dev/null +++ b/travel_times/baseline.csv @@ -0,0 +1,14 @@ +label,estimated_minutes +総武快速線 快速 錦糸町→津田沼,22.4690 +総武快速線 快速 新小岩→津田沼,17.3157 +京成 スカイライナー 日暮里→空港第2ビル,42.1152 +京急本線 快特 品川→横浜,17.6754 +小田急小田原線 快速急行 新宿→町田,32.8432 +東急東横線 特急 渋谷→横浜,26.4775 +京王井の頭線 急行 渋谷→吉祥寺,16.8168 +京王線 特急 新宿→京王八王子,41.0112 +阪急神戸本線 特急 大阪梅田→神戸三宮,26.6167 +西武池袋線 準急 池袋→所沢,29.4847 +西武池袋線 急行 池袋→所沢,22.4978 +東京メトロ銀座線 渋谷→新橋,13.2494 +つくばエクスプレス 普通 秋葉原→つくば,61.8611 diff --git a/travel_times/cases.csv b/travel_times/cases.csv new file mode 100644 index 00000000..cc76e78d --- /dev/null +++ b/travel_times/cases.csv @@ -0,0 +1,19 @@ +label,line_group_id,from_station_id,to_station_id,slice_end_station_id,measure,real_min_minutes,real_max_minutes,source +総武快速線 快速 錦糸町→津田沼,40,1131404,1131408,1131410,departure,19,25,メンテナ確認 +総武快速線 快速 新小岩→津田沼,40,1131405,1131408,1131410,departure,15,20,メンテナ確認 +京成 スカイライナー 日暮里→空港第2ビル,197,2300102,2300610,,arrival,40,48,メンテナ確認 +京急本線 快特 品川→横浜,74,2700102,2700126,,arrival,17,22,メンテナ確認 +小田急小田原線 快速急行 新宿→町田,129,2500101,2500127,,arrival,30,31,メンテナ確認 +東急東横線 特急 渋谷→横浜,161,2600101,2600121,,arrival,27,28,メンテナ確認 +京王井の頭線 急行 渋谷→吉祥寺,72,2400601,2400617,,arrival,17,22,メンテナ確認 +京王線 特急 新宿→京王八王子,71,2400101,2400134,,arrival,45,45,メンテナ確認 +阪急神戸本線 特急 大阪梅田→神戸三宮,384,3400101,3400116,,arrival,27,28,メンテナ確認 +西武池袋線 準急 池袋→所沢,132,2200101,2200131,,arrival,30,38,メンテナ確認 +西武池袋線 急行 池袋→所沢,190,2200101,2200131,,arrival,21,25,メンテナ確認 +都営大江戸線 落合南長崎→光が丘,1000099301,9930133,9930138,,arrival,11,12,東京都交通局 GTFS +都営大江戸線 新宿→光が丘,1000099301,9930128,9930138,,arrival,24,26,東京都交通局 GTFS +都営大江戸線 光が丘→都庁前,1000099301,9930138,9930101,,arrival,21,21,東京都交通局 GTFS +都営大江戸線 清澄白河→赤羽橋,1000099301,9930115,9930122,,arrival,16,17,東京都交通局 GTFS +都営大江戸線 光が丘→都庁前 (環状部経由),1000099301,9930138,9930100,,arrival,84,86,東京都交通局 GTFS +東京メトロ銀座線 渋谷→新橋,1137,2800119,2800112,,arrival,15,15,東京メトロ GTFS +つくばエクスプレス 普通 秋葉原→つくば,977,9930901,9930920,,arrival,63,66,首都圏新都市鉄道 GTFS From df03bd6f19db1546673846a8a759d74e19035584 Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Thu, 1 Oct 2026 02:20:34 +0900 Subject: [PATCH 2/4] =?UTF-8?q?=E6=89=80=E8=A6=81=E6=99=82=E9=96=93?= =?UTF-8?q?=E3=83=99=E3=83=B3=E3=83=81=E3=83=9E=E3=83=BC=E3=82=AF=E3=81=A7?= =?UTF-8?q?=E8=A8=98=E9=8C=B2=E6=B8=88=E3=81=BF=E3=81=AE=E5=9F=BA=E6=BA=96?= =?UTF-8?q?=E3=82=92=E6=8E=A8=E5=AE=9A=E3=81=A7=E3=81=8D=E3=81=AA=E3=81=8F?= =?UTF-8?q?=E3=81=AA=E3=81=A3=E3=81=9F=E3=82=89=E5=A4=B1=E6=95=97=E3=81=95?= =?UTF-8?q?=E3=81=9B=E3=80=81generated=E3=81=AE=E3=83=87=E3=83=BC=E3=82=BF?= =?UTF-8?q?=E3=81=A7=E3=81=AF=E8=A8=98=E9=8C=B2=E3=81=A8=E6=AF=94=E3=81=B9?= =?UTF-8?q?=E3=81=AA=E3=81=84=E3=82=88=E3=81=86=E3=81=AB=E3=81=97=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 --- Makefile | 2 +- build.rs | 10 ++++++++++ src/travel_times.rs | 28 +++++++++++++++++++++++++++- travel_times/README.md | 7 ++++++- 4 files changed, 44 insertions(+), 3 deletions(-) diff --git a/Makefile b/Makefile index da433b6c..f0153e94 100644 --- a/Makefile +++ b/Makefile @@ -90,7 +90,7 @@ ipa-audit: # ステージングに向ける (TRAVEL_TIME_API)。 TRAVEL_TIME_API ?= http://127.0.0.1:8787/ travel-time-report: - python3 scripts/travel_time_report.py --api $(TRAVEL_TIME_API) + python3 scripts/travel_time_report.py --api "$(TRAVEL_TIME_API)" bench: @echo "警告: 本番 (gql.trainlcd.app) とステージングへ実リクエストを送ります。" >&2 diff --git a/build.rs b/build.rs index 4fb2a683..e9bff096 100644 --- a/build.rs +++ b/build.rs @@ -85,6 +85,16 @@ fn main() { staged.len() ); } + // どちらのデータを埋め込んだか。所要時間のベンチマーク (src/travel_times.rs) の + // 記録は data/*.csv で作るので、生成データのときは記録と比べない + println!( + "cargo:rustc-env=STATIONAPI_EMBEDDED_DATA={}", + if generated_count == 0 { + "data" + } else { + "generated" + } + ); if generated_count == 0 { println!( "cargo:warning=generated が無いため data/*.csv を使用します。\ diff --git a/src/travel_times.rs b/src/travel_times.rs index 0ddf4945..f60e9e17 100644 --- a/src/travel_times.rs +++ b/src/travel_times.rs @@ -211,7 +211,15 @@ fn estimates_do_not_drift_away_from_real_travel_times() { } println!("{}", report.join("\n")); + // 記録は CI と同じ data/*.csv で作る。`make data` で生成データを置いた手元では + // 推定できる基準も推定の値も変わるので、記録とは比べない + let embedded_data_only = env!("STATIONAPI_EMBEDDED_DATA") == "data"; + if std::env::var("TRAVEL_TIMES_UPDATE_BASELINE").as_deref() == Ok("1") { + assert!( + embedded_data_only, + "記録は data/*.csv で作る。generated/ を退けてから更新する" + ); let mut out = String::from("label,estimated_minutes\n"); for (case, est) in &estimated { if let Some(est) = est { @@ -222,11 +230,29 @@ fn estimates_do_not_drift_away_from_real_travel_times() { return; } + if !embedded_data_only { + println!( + "generated/ のデータで動いているので記録とは比べない。\ + 本番と同じデータでの精度は make travel-time-report で測る" + ); + return; + } + let baseline = read_baseline(); let (mut now_sum, mut base_sum, mut n) = (0.0, 0.0, 0); let mut worse = Vec::new(); for (case, est) in &estimated { - let Some(est) = est else { continue }; + // 生成データにしか無い種別グループの基準は記録に入らないので飛ばしてよい。 + // 記録済みの基準を推定できなくなったのは、データの変更で種別グループが + // 消えたなどの異常なので、黙って比較から外さずに止める + let Some(est) = est else { + assert!( + !baseline.contains_key(&case.label), + "{}: 記録済みの基準を推定できない (種別グループがデータから消えた可能性)", + case.label + ); + continue; + }; let base = baseline.get(&case.label).unwrap_or_else(|| { panic!( "{}: 記録が無い。TRAVEL_TIMES_UPDATE_BASELINE=1 で記録を更新する", diff --git a/travel_times/README.md b/travel_times/README.md index 7d81d0fb..06ab20e6 100644 --- a/travel_times/README.md +++ b/travel_times/README.md @@ -44,7 +44,12 @@ 外れが 1 ポイントより多く増えるか、平均が悪くなると失敗します。 CI は `generated/` を作らずに `data/*.csv` で動くので、生成データにしか無い種別 -グループ (各駅停車を補う系統など) の基準は飛ばします。 +グループ (各駅停車を補う系統など) の基準は飛ばします。記録にある基準を推定できなく +なったとき (種別グループがデータから消えたときなど) は、飛ばさずに失敗します。 + +記録は `data/*.csv` で作ります。`make data` で `generated/` を置いた手元では、 +推定できる基準も推定の値も変わるので、記録とは比べずに表だけを出します。記録の +更新も `generated/` があると止まります。 推定を意図して変えたときは、記録を更新して同じ PR に含めます。 From 7536d24a44810c1a4ab2102173072e21206f8076 Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Thu, 1 Oct 2026 02:40:19 +0900 Subject: [PATCH 3/4] =?UTF-8?q?=E6=89=80=E8=A6=81=E6=99=82=E9=96=93?= =?UTF-8?q?=E3=83=99=E3=83=B3=E3=83=81=E3=83=9E=E3=83=BC=E3=82=AF=E3=81=AE?= =?UTF-8?q?=E8=A8=98=E9=8C=B2=E3=82=92=E7=94=9F=E6=88=90=E3=83=87=E3=83=BC?= =?UTF-8?q?=E3=82=BF=E3=81=A7=E4=BD=9C=E3=82=8A=E3=80=81build=5Fworker.yml?= =?UTF-8?q?=E3=81=A7=E7=94=9F=E6=88=90=E3=83=87=E3=83=BC=E3=82=BF=E3=82=92?= =?UTF-8?q?=E4=BD=9C=E3=81=A3=E3=81=A6=E3=81=8B=E3=82=89=E6=AF=94=E3=81=B9?= =?UTF-8?q?=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB=E3=81=97=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 --- .github/workflows/build_worker.yml | 9 +++++++++ AGENTS.md | 2 +- build.rs | 2 +- docs/architecture.md | 3 ++- src/travel_times.rs | 32 ++++++++++++++++-------------- travel_times/README.md | 20 +++++++++---------- travel_times/baseline.csv | 5 +++++ 7 files changed, 45 insertions(+), 28 deletions(-) diff --git a/.github/workflows/build_worker.yml b/.github/workflows/build_worker.yml index 2a7b2f7d..fa92c431 100644 --- a/.github/workflows/build_worker.yml +++ b/.github/workflows/build_worker.yml @@ -23,6 +23,7 @@ on: - "Cargo.toml" - ".github/actions/build-worker/action.yml" - ".github/workflows/build_worker.yml" + - "travel_times/**" push: # dev / master は deploy_staging.yml / deploy_production.yml が同じ # composite action で検証してからデプロイするため、ここでは走らせない。 @@ -42,6 +43,7 @@ on: - "Cargo.toml" - ".github/actions/build-worker/action.yml" - ".github/workflows/build_worker.yml" + - "travel_times/**" workflow_dispatch: name: Build Cloudflare Worker @@ -70,6 +72,13 @@ jobs: - uses: ./.github/actions/build-worker + # 到着時間推定の所要時間の見張り (src/travel_times.rs)。記録 + # (travel_times/baseline.csv) は本番と同じ生成データで作るので、上の + # action が generated/ を作ったこのジョブで比べる。バスのフィードが + # 欠けても鉄道の推定は変わらない。 + - name: Check travel times against the benchmark + run: cargo test -p stationapi-worker travel_times + - uses: actions/upload-artifact@v4 with: name: worker-build diff --git a/AGENTS.md b/AGENTS.md index 9dd2688f..0cfd2998 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -64,7 +64,7 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - **Schema** – Changing a GraphQL type changes the SDL. Update `schema/public.graphql` in the same change; CI compares it against the running Worker's `/__schema` and fails on any difference. That diff is exactly the client-visible impact. - **Data verification** – Execute `cargo run -p data_validator` whenever CSVs change and record results in pull requests. - **IPA coverage audit** – Execute `make ipa-audit` when English or romanized CSV names change. This is a read-only report for `data/2!lines.csv`, `data/3!stations.csv`, and `data/4!types.csv`; it does not fail validation, but highlights unresolved tokens and example names so the IPA dictionary can be extended deliberately. -- **Travel-time benchmark** – `travel_times/cases.csv` lists real travel times (a range in minutes, weekday daytime, trains with the same stopping pattern as the line group) that the arrival estimation is measured against. `cargo test -p stationapi-worker` (`src/travel_times.rs`) fails when any case moves more than 1 percentage point further outside its range than `travel_times/baseline.csv` records, or when the mean gets worse; CI runs on `data/*.csv`, so cases whose line group exists only in `generated/` are skipped. `make travel-time-report` (`TRAVEL_TIME_API`, default the local `make dev` Worker) measures every case against a Worker built from the generated data. A change to the speed tables, the general speed rules, or the estimation parameters must attach the report from before and after, and must not trade one line's accuracy for the whole set's; after an intended change, regenerate the baseline with `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times`. Values come from open GTFS feeds (credit them in the README's Data Sources) or from the maintainer. +- **Travel-time benchmark** – `travel_times/cases.csv` lists real travel times (a range in minutes, weekday daytime, trains with the same stopping pattern as the line group) that the arrival estimation is measured against. `cargo test -p stationapi-worker` (`src/travel_times.rs`) fails when any case moves more than 1 percentage point further outside its range than `travel_times/baseline.csv` records, when the mean gets worse, or when a recorded case can no longer be estimated. It compares only when the Worker embeds `generated/` (the estimation uses track lengths and line groups that exist only there), so `build_worker.yml` runs it after building the data; on `data/*.csv` it just prints the table. `make travel-time-report` (`TRAVEL_TIME_API`, default the local `make dev` Worker) measures every case against a Worker built from the generated data. A change to the speed tables, the general speed rules, or the estimation parameters must attach the report from before and after, and must not trade one line's accuracy for the whole set's; after an intended change, run `make data` and regenerate the baseline with `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times`. Values come from open GTFS feeds (credit them in the README's Data Sources) or from the maintainer. - **Endpoint benchmarks** – `make bench` (or `python3 .claude/skills/benchmark-gql/bench.py`) replays every `Query` field against production (`gql.trainlcd.app`, script `stationapi`) and staging (`gql-stg.trainlcd.app`, script `stationapi-stg`) and writes a Markdown report under `benchmarks/`. Both environments embed the same data, so any difference is implementation — which makes this the way to see what a `dev`-to-`master` release will do to performance before it ships. Besides client latency it records the Worker's `cpuTime`, read from `wrangler tail --format json` and matched to each request by `cf-ray`; the tail is filtered on a per-run request header, so production's live traffic does not leak into the sample. Collecting CPU time needs the `workers_tail (read)` scope, and the run sends hundreds of real requests to production — it is not a routine check. Add a case to `.claude/skills/benchmark-gql/queries.json` whenever a `Query` field is added, and never edit an existing case's variables: the reports are meant to stay comparable across runs. ## GraphQL Query Overview diff --git a/build.rs b/build.rs index e9bff096..2cdc543d 100644 --- a/build.rs +++ b/build.rs @@ -86,7 +86,7 @@ fn main() { ); } // どちらのデータを埋め込んだか。所要時間のベンチマーク (src/travel_times.rs) の - // 記録は data/*.csv で作るので、生成データのときは記録と比べない + // 記録は生成データで作るので、data/*.csv のときは記録と比べない println!( "cargo:rustc-env=STATIONAPI_EMBEDDED_DATA={}", if generated_count == 0 { diff --git a/docs/architecture.md b/docs/architecture.md index 3d1eb55b..27727ff6 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -515,7 +515,8 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int! - `cargo test -p stationapi-worker` (`src/travel_times.rs`) は、基準ごとの「実際の範囲 からの外れ」を `travel_times/baseline.csv` の記録と比べ、悪くなると失敗します。 - CI は `data/*.csv` で動くので、生成データにしか無い種別グループの基準は飛ばします。 + 記録は本番と同じ生成データで作るので、比べるのは `generated/` で動くときだけです。 + CI では `build_worker.yml` が生成データを作ってから走らせます。 - `make travel-time-report` は、生成データで動く Worker に問い合わせて全件の誤差を 出します。推定の規則や較正を変える PR には、変更前と変更後のレポートを載せます。 diff --git a/src/travel_times.rs b/src/travel_times.rs index f60e9e17..873511d7 100644 --- a/src/travel_times.rs +++ b/src/travel_times.rs @@ -5,11 +5,13 @@ //! 悪くなった基準があるか、平均が悪くなったら失敗にする。速度の較正や一般則を //! 1 つの路線に合わせて変えたときに、ほかの路線がどれだけ崩れたかをここで見る。 //! -//! CI は `generated/` を作らずに `data/*.csv` で動くので、生成データにしか無い -//! 種別グループ (各駅停車を補う系統など) の基準は飛ばす。本番と同じデータでの -//! 精度は `make travel-time-report` で測る。 +//! 記録は、本番と同じ生成データ (`make data` で作る `generated/`) で出した推定で、 +//! 比べるのも生成データのときだけにする。到着時間推定は、生成データにしか無い +//! 線路の長さや種別グループを使うので、`data/*.csv` だけでは本番の推定を再現 +//! できない。CI では `build_worker.yml` が `generated/` を作ってからこれを走らせる。 +//! `data/*.csv` で動くとき (`ci.yml` のテストなど) は、表を出すだけにする。 //! -//! 推定を意図して変えたときは、次で記録を更新する。 +//! 推定を意図して変えたときは、`make data` のあとに次で記録を更新する。 //! `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times` use std::collections::HashMap; @@ -211,14 +213,14 @@ fn estimates_do_not_drift_away_from_real_travel_times() { } println!("{}", report.join("\n")); - // 記録は CI と同じ data/*.csv で作る。`make data` で生成データを置いた手元では - // 推定できる基準も推定の値も変わるので、記録とは比べない - let embedded_data_only = env!("STATIONAPI_EMBEDDED_DATA") == "data"; + // 記録は本番と同じ生成データで作る。data/*.csv では線路の長さや生成された + // 種別グループが無く、推定が本番と違うので、記録とは比べない + let embedded_generated = env!("STATIONAPI_EMBEDDED_DATA") == "generated"; if std::env::var("TRAVEL_TIMES_UPDATE_BASELINE").as_deref() == Ok("1") { assert!( - embedded_data_only, - "記録は data/*.csv で作る。generated/ を退けてから更新する" + embedded_generated, + "記録は生成データで作る。make data で generated/ を作ってから更新する" ); let mut out = String::from("label,estimated_minutes\n"); for (case, est) in &estimated { @@ -230,10 +232,10 @@ fn estimates_do_not_drift_away_from_real_travel_times() { return; } - if !embedded_data_only { + if !embedded_generated { println!( - "generated/ のデータで動いているので記録とは比べない。\ - 本番と同じデータでの精度は make travel-time-report で測る" + "data/*.csv で動いているので記録とは比べない。\ + make data で generated/ を作ると記録と比べる" ); return; } @@ -242,9 +244,9 @@ fn estimates_do_not_drift_away_from_real_travel_times() { let (mut now_sum, mut base_sum, mut n) = (0.0, 0.0, 0); let mut worse = Vec::new(); for (case, est) in &estimated { - // 生成データにしか無い種別グループの基準は記録に入らないので飛ばしてよい。 - // 記録済みの基準を推定できなくなったのは、データの変更で種別グループが - // 消えたなどの異常なので、黙って比較から外さずに止める + // 推定できない基準は記録にも入らないので飛ばしてよい。記録済みの基準を + // 推定できなくなったのは、データの変更で種別グループが消えたなどの異常 + // なので、黙って比較から外さずに止める let Some(est) = est else { assert!( !baseline.contains_key(&case.label), diff --git a/travel_times/README.md b/travel_times/README.md index 06ab20e6..b212d639 100644 --- a/travel_times/README.md +++ b/travel_times/README.md @@ -10,7 +10,7 @@ | パス | 内容 | | --- | --- | | `cases.csv` | 基準の一覧。1 行が 1 区間 | -| `baseline.csv` | CI のデータ (`data/*.csv`) で出した推定の所要時間の記録 | +| `baseline.csv` | 本番と同じ生成データ (`make data` で作る `generated/`) で出した推定の所要時間の記録 | `cases.csv` の列は次のとおりです。 @@ -41,19 +41,19 @@ `cargo test -p stationapi-worker` の `travel_times` モジュール (`src/travel_times.rs`) が、基準ごとに推定を出し、実際の範囲からの外れを求めます。範囲に入っていれば 0、 外れていれば近い端に対する割合です。`baseline.csv` の記録と比べて、1 件でも -外れが 1 ポイントより多く増えるか、平均が悪くなると失敗します。 +外れが 1 ポイントより多く増えるか、平均が悪くなると失敗します。記録にある基準を +推定できなくなったとき (種別グループがデータから消えたときなど) も失敗します。 -CI は `generated/` を作らずに `data/*.csv` で動くので、生成データにしか無い種別 -グループ (各駅停車を補う系統など) の基準は飛ばします。記録にある基準を推定できなく -なったとき (種別グループがデータから消えたときなど) は、飛ばさずに失敗します。 +比べるのは、本番と同じ生成データ (`generated/`) で動くときだけです。到着時間推定は、 +生成データにしか無い線路の長さや種別グループを使うので、`data/*.csv` だけでは本番の +推定を再現できません。CI では `build_worker.yml` が `generated/` を作ってから走らせ +ます。`data/*.csv` で動くとき (`ci.yml` のテストなど) は、表を出すだけにします。 -記録は `data/*.csv` で作ります。`make data` で `generated/` を置いた手元では、 -推定できる基準も推定の値も変わるので、記録とは比べずに表だけを出します。記録の -更新も `generated/` があると止まります。 - -推定を意図して変えたときは、記録を更新して同じ PR に含めます。 +推定を意図して変えたときは、`make data` のあとに記録を更新して、同じ PR に含めます。 +`generated/` が無いと更新は止まります。 ```bash +make data TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times ``` diff --git a/travel_times/baseline.csv b/travel_times/baseline.csv index de4a0b93..0bfaf088 100644 --- a/travel_times/baseline.csv +++ b/travel_times/baseline.csv @@ -10,5 +10,10 @@ label,estimated_minutes 阪急神戸本線 特急 大阪梅田→神戸三宮,26.6167 西武池袋線 準急 池袋→所沢,29.4847 西武池袋線 急行 池袋→所沢,22.4978 +都営大江戸線 落合南長崎→光が丘,11.5568 +都営大江戸線 新宿→光が丘,24.8605 +都営大江戸線 光が丘→都庁前,22.6198 +都営大江戸線 清澄白河→赤羽橋,16.0367 +都営大江戸線 光が丘→都庁前 (環状部経由),84.3894 東京メトロ銀座線 渋谷→新橋,13.2494 つくばエクスプレス 普通 秋葉原→つくば,61.8611 From 5efd16cfb5bc0cf276cecdf05bbfb18c38d82d98 Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Thu, 1 Oct 2026 03:04:00 +0900 Subject: [PATCH 4/4] =?UTF-8?q?=E6=89=80=E8=A6=81=E6=99=82=E9=96=93?= =?UTF-8?q?=E3=83=99=E3=83=B3=E3=83=81=E3=83=9E=E3=83=BC=E3=82=AF=E3=81=AB?= =?UTF-8?q?=E5=85=B8=E5=9E=8B=E7=9A=84=E3=81=AA=E6=89=80=E8=A6=81=E6=99=82?= =?UTF-8?q?=E9=96=93=E3=81=AE=E5=88=97=E3=82=92=E8=B6=B3=E3=81=97=E3=80=81?= =?UTF-8?q?=E5=9B=9E=E5=B8=B0=E3=81=AE=E8=A6=8B=E5=BC=B5=E3=82=8A=E3=82=92?= =?UTF-8?q?=E7=AF=84=E5=9B=B2=E3=81=8B=E3=82=89=E3=81=AE=E5=A4=96=E3=82=8C?= =?UTF-8?q?=E3=81=A7=E3=81=AF=E3=81=AA=E3=81=8F=E5=85=B8=E5=9E=8B=E5=80=A4?= =?UTF-8?q?=E3=81=8B=E3=82=89=E3=81=AE=E3=81=9A=E3=82=8C=E3=81=A7=E5=88=A4?= =?UTF-8?q?=E5=AE=9A=E3=81=99=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB=E3=81=97?= =?UTF-8?q?=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 5.5 --- AGENTS.md | 2 +- docs/architecture.md | 7 ++--- scripts/travel_time_report.py | 22 +++++++++------- src/travel_times.rs | 49 +++++++++++++++++++++++++---------- travel_times/README.md | 12 ++++++--- travel_times/cases.csv | 38 +++++++++++++-------------- 6 files changed, 80 insertions(+), 50 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 0cfd2998..50795c94 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -64,7 +64,7 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - **Schema** – Changing a GraphQL type changes the SDL. Update `schema/public.graphql` in the same change; CI compares it against the running Worker's `/__schema` and fails on any difference. That diff is exactly the client-visible impact. - **Data verification** – Execute `cargo run -p data_validator` whenever CSVs change and record results in pull requests. - **IPA coverage audit** – Execute `make ipa-audit` when English or romanized CSV names change. This is a read-only report for `data/2!lines.csv`, `data/3!stations.csv`, and `data/4!types.csv`; it does not fail validation, but highlights unresolved tokens and example names so the IPA dictionary can be extended deliberately. -- **Travel-time benchmark** – `travel_times/cases.csv` lists real travel times (a range in minutes, weekday daytime, trains with the same stopping pattern as the line group) that the arrival estimation is measured against. `cargo test -p stationapi-worker` (`src/travel_times.rs`) fails when any case moves more than 1 percentage point further outside its range than `travel_times/baseline.csv` records, when the mean gets worse, or when a recorded case can no longer be estimated. It compares only when the Worker embeds `generated/` (the estimation uses track lengths and line groups that exist only there), so `build_worker.yml` runs it after building the data; on `data/*.csv` it just prints the table. `make travel-time-report` (`TRAVEL_TIME_API`, default the local `make dev` Worker) measures every case against a Worker built from the generated data. A change to the speed tables, the general speed rules, or the estimation parameters must attach the report from before and after, and must not trade one line's accuracy for the whole set's; after an intended change, run `make data` and regenerate the baseline with `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times`. Values come from open GTFS feeds (credit them in the README's Data Sources) or from the maintainer. +- **Travel-time benchmark** – `travel_times/cases.csv` lists real travel times (a range and a typical value — the median — in minutes, weekday daytime, trains with the same stopping pattern as the line group) that the arrival estimation is measured against. `cargo test -p stationapi-worker` (`src/travel_times.rs`) fails when any case moves more than 1 percentage point further from its typical value than `travel_times/baseline.csv` records (a range alone hides drifts inside a range widened by one outlier train), when the mean gets worse, or when a recorded case can no longer be estimated. It compares only when the Worker embeds `generated/` (the estimation uses track lengths and line groups that exist only there), so `build_worker.yml` runs it after building the data; on `data/*.csv` it just prints the table. `make travel-time-report` (`TRAVEL_TIME_API`, default the local `make dev` Worker) measures every case against a Worker built from the generated data. A change to the speed tables, the general speed rules, or the estimation parameters must attach the report from before and after, and must not trade one line's accuracy for the whole set's; after an intended change, run `make data` and regenerate the baseline with `TRAVEL_TIMES_UPDATE_BASELINE=1 cargo test -p stationapi-worker travel_times`. Values come from open GTFS feeds (credit them in the README's Data Sources) or from the maintainer. - **Endpoint benchmarks** – `make bench` (or `python3 .claude/skills/benchmark-gql/bench.py`) replays every `Query` field against production (`gql.trainlcd.app`, script `stationapi`) and staging (`gql-stg.trainlcd.app`, script `stationapi-stg`) and writes a Markdown report under `benchmarks/`. Both environments embed the same data, so any difference is implementation — which makes this the way to see what a `dev`-to-`master` release will do to performance before it ships. Besides client latency it records the Worker's `cpuTime`, read from `wrangler tail --format json` and matched to each request by `cf-ray`; the tail is filtered on a per-run request header, so production's live traffic does not leak into the sample. Collecting CPU time needs the `workers_tail (read)` scope, and the run sends hundreds of real requests to production — it is not a routine check. Add a case to `.claude/skills/benchmark-gql/queries.json` whenever a `Query` field is added, and never edit an existing case's variables: the reports are meant to stay comparable across runs. ## GraphQL Query Overview diff --git a/docs/architecture.md b/docs/architecture.md index 27727ff6..9ff2fd9f 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -513,9 +513,10 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int! 規則を使うほかの路線の推定も変わります。変更の前後で全体の誤差を測るための仕組み です。 -- `cargo test -p stationapi-worker` (`src/travel_times.rs`) は、基準ごとの「実際の範囲 - からの外れ」を `travel_times/baseline.csv` の記録と比べ、悪くなると失敗します。 - 記録は本番と同じ生成データで作るので、比べるのは `generated/` で動くときだけです。 +- `cargo test -p stationapi-worker` (`src/travel_times.rs`) は、基準ごとの「実際の + 典型的な所要時間からのずれ」を `travel_times/baseline.csv` の記録と比べ、悪くなる + と失敗します。記録は本番と同じ生成データで作るので、比べるのは `generated/` で + 動くときだけです。 CI では `build_worker.yml` が生成データを作ってから走らせます。 - `make travel-time-report` は、生成データで動く Worker に問い合わせて全件の誤差を 出します。推定の規則や較正を変える PR には、変更前と変更後のレポートを載せます。 diff --git a/scripts/travel_time_report.py b/scripts/travel_time_report.py index d4d0d049..dd9fc437 100644 --- a/scripts/travel_time_report.py +++ b/scripts/travel_time_report.py @@ -72,35 +72,37 @@ def main() -> int: with CASES.open(encoding="utf-8") as f: cases = list(csv.DictReader(f)) - rows, range_errors, mid_errors, failed = [], [], [], [] + rows, range_errors, typical_errors, failed = [], [], [], [] for case in cases: lo, hi = float(case["real_min_minutes"]), float(case["real_max_minutes"]) + typical = float(case["real_typical_minutes"]) try: est = estimate(args.api, case) except Exception as e: # noqa: BLE001 - 1 件の失敗で全体を止めない failed.append(f"{case['label']}: {e}") continue - mid = (lo + hi) / 2 - r, m = range_error(est, lo, hi), (est - mid) / mid + t, r = (est - typical) / typical, range_error(est, lo, hi) + typical_errors.append(abs(t)) range_errors.append(r) - mid_errors.append(abs(m)) rows.append( - f"| {case['label']} | {lo:g}〜{hi:g}分 | {est:.1f}分 | {r * 100:.1f}% | {m * 100:+.1f}% |" + f"| {case['label']} | {typical:g}分 ({lo:g}〜{hi:g}分) | {est:.1f}分 " + f"| {t * 100:+.1f}% | {r * 100:.1f}% |" ) print(f"# 到着時間推定の誤差 ({args.api})\n") - print("| 基準 | 実際 | 推定 | 範囲からの外れ | 範囲の中央からのずれ |") + print("| 基準 | 実際の典型 (範囲) | 推定 | 典型からのずれ | 範囲からの外れ |") print("| --- | --- | --- | --- | --- |") print("\n".join(rows)) - if range_errors: + if typical_errors: print( - f"\n{len(range_errors)} 件: 範囲からの外れの平均 {statistics.mean(range_errors) * 100:.2f}%、" - f"範囲の中央からのずれ (絶対値) の平均 {statistics.mean(mid_errors) * 100:.2f}%" + f"\n{len(typical_errors)} 件: 典型からのずれ (絶対値) の平均 " + f"{statistics.mean(typical_errors) * 100:.2f}%、" + f"範囲からの外れの平均 {statistics.mean(range_errors) * 100:.2f}%" ) if failed: print("\n推定できなかった基準:\n") print("\n".join(f"- {line}" for line in failed)) - return 0 if range_errors else 1 + return 0 if typical_errors else 1 if __name__ == "__main__": diff --git a/src/travel_times.rs b/src/travel_times.rs index 873511d7..1c6aeff3 100644 --- a/src/travel_times.rs +++ b/src/travel_times.rs @@ -1,9 +1,13 @@ //! 実際の所要時間 (`travel_times/cases.csv`) に対する到着時間推定の回帰の見張り。 //! -//! 基準ごとに推定の所要時間を出し、実際の範囲からの外れ (範囲内なら 0、外れたら -//! 近い端からの割合) を求める。記録した推定 (`travel_times/baseline.csv`) より -//! 悪くなった基準があるか、平均が悪くなったら失敗にする。速度の較正や一般則を -//! 1 つの路線に合わせて変えたときに、ほかの路線がどれだけ崩れたかをここで見る。 +//! 基準ごとに推定の所要時間を出し、実際の典型的な所要時間 (平日日中の中央値) から +//! のずれを求める。記録した推定 (`travel_times/baseline.csv`) より悪くなった基準が +//! あるか、平均が悪くなったら失敗にする。速度の較正や一般則を 1 つの路線に合わせて +//! 変えたときに、ほかの路線がどれだけ崩れたかをここで見る。 +//! +//! 実際の所要時間の範囲 (最小〜最大) からの外れも表に出すが、判定には使わない。 +//! 待ち合わせなどで範囲に外れ値の列車が入ると幅が広がり、範囲内なら誤差 0 と +//! 数える物差しでは、典型的な値から大きく離れても見逃すため。 //! //! 記録は、本番と同じ生成データ (`make data` で作る `generated/`) で出した推定で、 //! 比べるのも生成データのときだけにする。到着時間推定は、生成データにしか無い @@ -24,7 +28,7 @@ use crate::repository::MemStationRepository; const CASES: &str = include_str!("../travel_times/cases.csv"); const BASELINE_PATH: &str = concat!(env!("CARGO_MANIFEST_DIR"), "/travel_times/baseline.csv"); -/// 1 基準あたり、範囲からの外れがこれだけ増えたら失敗にする (割合) +/// 1 基準あたり、典型的な値からのずれがこれだけ増えたら失敗にする (割合) const CASE_TOLERANCE: f64 = 0.01; /// 平均の比較の許容幅 (割合)。記録は推定の分を小数 4 桁に丸めて書くので、その丸めで /// 平均がわずかに動くぶんを吸収する @@ -50,9 +54,16 @@ struct Case { measure: Measure, real_min: f64, real_max: f64, + /// 典型的な所要時間 (平日日中の中央値。分からないときは範囲の中央) + real_typical: f64, } impl Case { + /// 典型的な所要時間からのずれ (絶対値の割合)。判定に使う指標 + fn typical_error(&self, estimated: f64) -> f64 { + (estimated - self.real_typical).abs() / self.real_typical + } + /// 実際の範囲からの外れ。範囲内なら 0、外れたら近い端に対する割合 fn range_error(&self, estimated: f64) -> f64 { if estimated < self.real_min { @@ -74,7 +85,7 @@ fn parse_cases() -> Vec { .position(|h| *h == name) .unwrap_or_else(|| panic!("列 {name} が無い")) }; - let (label, group, from, to, end, measure, min, max) = ( + let (label, group, from, to, end, measure, min, max, typical) = ( col("label"), col("line_group_id"), col("from_station_id"), @@ -83,6 +94,7 @@ fn parse_cases() -> Vec { col("measure"), col("real_min_minutes"), col("real_max_minutes"), + col("real_typical_minutes"), ); lines .filter(|line| !line.trim().is_empty()) @@ -116,6 +128,7 @@ fn parse_cases() -> Vec { measure, real_min: minutes(min), real_max: minutes(max), + real_typical: minutes(typical), } }) .collect() @@ -177,7 +190,9 @@ fn cases_are_well_formed() { for case in &cases { assert!(labels.insert(&case.label), "label が重複: {}", case.label); assert!( - case.real_min > 0.0 && case.real_min <= case.real_max, + case.real_min > 0.0 + && case.real_min <= case.real_typical + && case.real_typical <= case.real_max, "{}", case.label ); @@ -196,17 +211,23 @@ fn estimates_do_not_drift_away_from_real_travel_times() { let cases = parse_cases(); let estimated: Vec<(&Case, Option)> = cases.iter().map(|c| (c, estimate(c))).collect(); - let mut report = vec![String::from("\n基準 | 実際 | 推定 | 範囲からの外れ")]; + let mut report = vec![String::from( + "\n基準 | 実際 (典型) | 推定 | 典型からのずれ | 範囲からの外れ", + )]; for (case, est) in &estimated { - let real = format!("{}〜{}分", case.real_min, case.real_max); + let real = format!( + "{}〜{}分 ({}分)", + case.real_min, case.real_max, case.real_typical + ); report.push(match est { Some(est) => format!( - "{} | {real} | {est:.1}分 | {:.1}%", + "{} | {real} | {est:.1}分 | {:.1}% | {:.1}%", case.label, + case.typical_error(*est) * 100.0, case.range_error(*est) * 100.0 ), None => format!( - "{} | {real} | (このデータに種別グループが無い) | -", + "{} | {real} | (このデータに種別グループが無い) | - | -", case.label ), }); @@ -261,10 +282,10 @@ fn estimates_do_not_drift_away_from_real_travel_times() { case.label ) }); - let (now_err, base_err) = (case.range_error(*est), case.range_error(*base)); + let (now_err, base_err) = (case.typical_error(*est), case.typical_error(*base)); if now_err > base_err + CASE_TOLERANCE { worse.push(format!( - "{}: 範囲からの外れが {:.1}% → {:.1}% (推定 {base:.1}分 → {est:.1}分)", + "{}: 典型的な所要時間からのずれが {:.1}% → {:.1}% (推定 {base:.1}分 → {est:.1}分)", case.label, base_err * 100.0, now_err * 100.0 @@ -283,7 +304,7 @@ fn estimates_do_not_drift_away_from_real_travel_times() { let (now_mean, base_mean) = (now_sum / n as f64, base_sum / n as f64); assert!( now_mean <= base_mean + MEAN_TOLERANCE, - "範囲からの外れの平均が {:.2}% → {:.2}% に悪化した", + "典型的な所要時間からのずれの平均が {:.2}% → {:.2}% に悪化した", base_mean * 100.0, now_mean * 100.0 ); diff --git a/travel_times/README.md b/travel_times/README.md index b212d639..f902f420 100644 --- a/travel_times/README.md +++ b/travel_times/README.md @@ -22,12 +22,17 @@ | `slice_end_station_id` | 推定する区間の終わり。`measure` が `departure` のときに、到着駅より先の駅を書く。空なら到着駅 | | `measure` | `arrival` は出発駅の発車から到着駅の到着まで。`departure` は到着駅の発車までで、到着時刻を載せない時刻表の値に使う | | `real_min_minutes` / `real_max_minutes` | 実際の所要時間の範囲 (分) | +| `real_typical_minutes` | 実際の典型的な所要時間 (分)。回帰の見張りはこの値からのずれで判定する | | `source` | 値の出どころ | ## 基準の決め方 - 平日の日中に出発駅を出る列車の所要時間を使います。途中で待ち合わせる列車などで 所要時間に幅があるときは、最小と最大をそのまま書きます。 +- 典型的な所要時間には、同じ列車の所要時間の中央値を書きます。中央値が分からず + 範囲だけが分かっているときは範囲の中央を書き、`source` にそう書き添えます。 + 範囲は外れ値の列車 1 本で広がるので、範囲に入っているかどうかだけでは、推定が + 典型的な値から離れたことを見逃します。 - 列車は、`line_group_id` の種別グループと同じ停車パターンのものに限ります。 停車駅が違う列車の所要時間と比べると、推定の誤差ではない差が混ざります。 - 出どころは、公開 GTFS か、メンテナが確認した値に限ります。GTFS から求めた値を @@ -39,9 +44,10 @@ ### CI (回帰の見張り) `cargo test -p stationapi-worker` の `travel_times` モジュール (`src/travel_times.rs`) -が、基準ごとに推定を出し、実際の範囲からの外れを求めます。範囲に入っていれば 0、 -外れていれば近い端に対する割合です。`baseline.csv` の記録と比べて、1 件でも -外れが 1 ポイントより多く増えるか、平均が悪くなると失敗します。記録にある基準を +が、基準ごとに推定を出し、典型的な所要時間からのずれ (絶対値の割合) を求めます。 +`baseline.csv` の記録と比べて、1 件でもずれが 1 ポイントより多く増えるか、平均が +悪くなると失敗します。範囲からの外れ (範囲内なら 0、外れたら近い端に対する割合) +も表に出しますが、判定には使いません。記録にある基準を 推定できなくなったとき (種別グループがデータから消えたときなど) も失敗します。 比べるのは、本番と同じ生成データ (`generated/`) で動くときだけです。到着時間推定は、 diff --git a/travel_times/cases.csv b/travel_times/cases.csv index cc76e78d..4f722854 100644 --- a/travel_times/cases.csv +++ b/travel_times/cases.csv @@ -1,19 +1,19 @@ -label,line_group_id,from_station_id,to_station_id,slice_end_station_id,measure,real_min_minutes,real_max_minutes,source -総武快速線 快速 錦糸町→津田沼,40,1131404,1131408,1131410,departure,19,25,メンテナ確認 -総武快速線 快速 新小岩→津田沼,40,1131405,1131408,1131410,departure,15,20,メンテナ確認 -京成 スカイライナー 日暮里→空港第2ビル,197,2300102,2300610,,arrival,40,48,メンテナ確認 -京急本線 快特 品川→横浜,74,2700102,2700126,,arrival,17,22,メンテナ確認 -小田急小田原線 快速急行 新宿→町田,129,2500101,2500127,,arrival,30,31,メンテナ確認 -東急東横線 特急 渋谷→横浜,161,2600101,2600121,,arrival,27,28,メンテナ確認 -京王井の頭線 急行 渋谷→吉祥寺,72,2400601,2400617,,arrival,17,22,メンテナ確認 -京王線 特急 新宿→京王八王子,71,2400101,2400134,,arrival,45,45,メンテナ確認 -阪急神戸本線 特急 大阪梅田→神戸三宮,384,3400101,3400116,,arrival,27,28,メンテナ確認 -西武池袋線 準急 池袋→所沢,132,2200101,2200131,,arrival,30,38,メンテナ確認 -西武池袋線 急行 池袋→所沢,190,2200101,2200131,,arrival,21,25,メンテナ確認 -都営大江戸線 落合南長崎→光が丘,1000099301,9930133,9930138,,arrival,11,12,東京都交通局 GTFS -都営大江戸線 新宿→光が丘,1000099301,9930128,9930138,,arrival,24,26,東京都交通局 GTFS -都営大江戸線 光が丘→都庁前,1000099301,9930138,9930101,,arrival,21,21,東京都交通局 GTFS -都営大江戸線 清澄白河→赤羽橋,1000099301,9930115,9930122,,arrival,16,17,東京都交通局 GTFS -都営大江戸線 光が丘→都庁前 (環状部経由),1000099301,9930138,9930100,,arrival,84,86,東京都交通局 GTFS -東京メトロ銀座線 渋谷→新橋,1137,2800119,2800112,,arrival,15,15,東京メトロ GTFS -つくばエクスプレス 普通 秋葉原→つくば,977,9930901,9930920,,arrival,63,66,首都圏新都市鉄道 GTFS +label,line_group_id,from_station_id,to_station_id,slice_end_station_id,measure,real_min_minutes,real_max_minutes,real_typical_minutes,source +総武快速線 快速 錦糸町→津田沼,40,1131404,1131408,1131410,departure,19,25,20,メンテナ確認 +総武快速線 快速 新小岩→津田沼,40,1131405,1131408,1131410,departure,15,20,16,メンテナ確認 +京成 スカイライナー 日暮里→空港第2ビル,197,2300102,2300610,,arrival,40,48,41,メンテナ確認 +京急本線 快特 品川→横浜,74,2700102,2700126,,arrival,17,22,17,メンテナ確認 +小田急小田原線 快速急行 新宿→町田,129,2500101,2500127,,arrival,30,31,30.5,メンテナ確認 (典型値は範囲の中央) +東急東横線 特急 渋谷→横浜,161,2600101,2600121,,arrival,27,28,27.5,メンテナ確認 (典型値は範囲の中央) +京王井の頭線 急行 渋谷→吉祥寺,72,2400601,2400617,,arrival,17,22,17,メンテナ確認 +京王線 特急 新宿→京王八王子,71,2400101,2400134,,arrival,45,45,45,メンテナ確認 +阪急神戸本線 特急 大阪梅田→神戸三宮,384,3400101,3400116,,arrival,27,28,27.5,メンテナ確認 (典型値は範囲の中央) +西武池袋線 準急 池袋→所沢,132,2200101,2200131,,arrival,30,38,38,メンテナ確認 +西武池袋線 急行 池袋→所沢,190,2200101,2200131,,arrival,21,25,24,メンテナ確認 +都営大江戸線 落合南長崎→光が丘,1000099301,9930133,9930138,,arrival,11,12,11,東京都交通局 GTFS +都営大江戸線 新宿→光が丘,1000099301,9930128,9930138,,arrival,24,26,24,東京都交通局 GTFS +都営大江戸線 光が丘→都庁前,1000099301,9930138,9930101,,arrival,21,21,21,東京都交通局 GTFS +都営大江戸線 清澄白河→赤羽橋,1000099301,9930115,9930122,,arrival,16,17,16,東京都交通局 GTFS +都営大江戸線 光が丘→都庁前 (環状部経由),1000099301,9930138,9930100,,arrival,84,86,84,東京都交通局 GTFS +東京メトロ銀座線 渋谷→新橋,1137,2800119,2800112,,arrival,15,15,15,東京メトロ GTFS +つくばエクスプレス 普通 秋葉原→つくば,977,9930901,9930920,,arrival,63,66,66,首都圏新都市鉄道 GTFS