Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions .github/workflows/build_worker.yml
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,7 @@ on:
- "Cargo.toml"
- ".github/actions/build-worker/action.yml"
- ".github/workflows/build_worker.yml"
- "travel_times/**"
push:
# dev / master は deploy_staging.yml / deploy_production.yml が同じ
# composite action で検証してからデプロイするため、ここでは走らせない。
Expand All @@ -42,6 +43,7 @@ on:
- "Cargo.toml"
- ".github/actions/build-worker/action.yml"
- ".github/workflows/build_worker.yml"
- "travel_times/**"
workflow_dispatch:

name: Build Cloudflare Worker
Expand Down Expand Up @@ -70,6 +72,13 @@ jobs:

- uses: ./.github/actions/build-worker

# 到着時間推定の所要時間の見張り (src/travel_times.rs)。記録
# (travel_times/baseline.csv) は本番と同じ生成データで作るので、上の
# action が generated/ を作ったこのジョブで比べる。バスのフィードが
# 欠けても鉄道の推定は変わらない。
- name: Check travel times against the benchmark
run: cargo test -p stationapi-worker travel_times

- uses: actions/upload-artifact@v4
with:
name: worker-build
Expand Down
7 changes: 4 additions & 3 deletions AGENTS.md

Large diffs are not rendered by default.

10 changes: 9 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
# StationAPI Makefile
# よく使うタスクの定義

.PHONY: help test check fmt clippy data build dev deploy deploy-production schema ipa-audit bench clean
.PHONY: help test check fmt clippy data build dev deploy deploy-production schema ipa-audit bench travel-time-report clean

# CI (.github/workflows/build_worker.yml) と同じ版を使う。グローバルへ入れて
# いなくても npx が取ってくるので、版ずれでビルド結果が変わらない。
Expand All @@ -22,6 +22,7 @@ help:
@echo " schema - Diff the running Worker's SDL against schema/public.graphql"
@echo " ipa-audit - Print IPA coverage report for English/romanized CSV names"
@echo " bench - Compare production vs staging GraphQL performance (sends live traffic to both)"
@echo " travel-time-report - Compare estimated travel times with travel_times/cases.csv (TRAVEL_TIME_API, default http://127.0.0.1:8787/)"
@echo " clean - Clean build artifacts"
@echo ""
@echo "Environment variables:"
Expand Down Expand Up @@ -84,6 +85,13 @@ ipa-audit:
# 実在のエンドポイントへ数百リクエスト投げるので、気軽に回すものではない。
# CPU Time の収集には wrangler の workers_tail (read) 権限が要る。
# 追加の引数は BENCH_ARGS で渡す (例: make bench BENCH_ARGS="--repeat 30")。
# 到着時間推定の所要時間を、実際の所要時間 (travel_times/cases.csv) と比べる。
# 本番と同じ生成データで測るため、`make data && make dev` で起動した Worker か
# ステージングに向ける (TRAVEL_TIME_API)。
TRAVEL_TIME_API ?= http://127.0.0.1:8787/
travel-time-report:
python3 scripts/travel_time_report.py --api "$(TRAVEL_TIME_API)"

bench:
@echo "警告: 本番 (gql.trainlcd.app) とステージングへ実リクエストを送ります。" >&2
@echo " 既定で 1 環境あたり 400 件超、うち数十件は Worker の CPU を 500ms 以上使います。" >&2
Expand Down
6 changes: 6 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,12 @@ This project includes a comprehensive dataset of Japanese railway information in
(https://nlftp.mlit.go.jp/ksj/gml/datalist/KsjTmplt-N02-2025.html) を加工して作成
- Bus stops and routes are derived from the GTFS and ODPT feeds listed in
`preprocessor/src/gtfs/feed.rs` and `preprocessor/src/gtfs/odpt.rs`.
- Some of the reference travel times in `travel_times/cases.csv` are derived from
the Toei Subway GTFS (Bureau of Transportation, Tokyo Metropolitan Government,
CC BY 4.0) and from the Tokyo Metro, Metropolitan Intercity Railway (Tsukuba
Express), Tokyo Waterfront Area Rapid Transit (Rinkai Line), and Tama Toshi
Monorail GTFS feeds published by the Public Transportation Open Data Center
under the Public Transportation Open Data Basic License.

## Contributors ✨

Expand Down
10 changes: 10 additions & 0 deletions build.rs
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,16 @@ fn main() {
staged.len()
);
}
// どちらのデータを埋め込んだか。所要時間のベンチマーク (src/travel_times.rs) の
// 記録は生成データで作るので、data/*.csv のときは記録と比べない
println!(
"cargo:rustc-env=STATIONAPI_EMBEDDED_DATA={}",
if generated_count == 0 {
"data"
} else {
"generated"
}
);
if generated_count == 0 {
println!(
"cargo:warning=generated が無いため data/*.csv を使用します。\
Expand Down
44 changes: 44 additions & 0 deletions docs/architecture.md
Original file line number Diff line number Diff line change
Expand Up @@ -475,6 +475,32 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int!
なので、各区間の最初の駅の `distanceFromPrevious` は 0 になり、通過駅が
あるか (優等種別の速度を使うか) も区間ごとに判定します。

`trainRoute` は、区間の値をどのモデルで出すかを `model: TrainRouteModel` で
選べます。

- `Legacy` (省略時): 追加した時点 (#1568) のモデルです。最高速度と加減速は
`dto::simulation::resolve_speed_profile` が決め、到着・出発の見込み
(`arrivalCumulativeMinutes` / `departureCumulativeMinutes`) は `null` です。
配布済みの MobileApp のオートモードがこの値で走るので、値を変えません。速度の較正は、
到着時間推定の較正を求め直す前の表を `domain/legacy_speed_table.rs` に凍結して
使います。
- `Estimated`: 到着時間推定 (`arrival_estimation`) のモデルで、MobileApp の
オートモードと GPX の生成が使います。返す駅列に推定を掛け、停車・通過、最高速度、
加減速を推定が使った値に置き換えて、到着・出発の見込みを入れます。見込みは、
同じ区間の `estimateArrivalTimes` と同じ値です (`legs` を渡したときは、同じ
`legs` を渡した `estimateArrivalTimes` と同じ値)。バスの駅を含む経路は推定の
モデルの対象外なので、`Legacy` と同じ値を返します。

2 つのモデルは、加減速、運転余裕率、停車時間、較正テーブル、駅間の距離が違い
ます。`Estimated` は、線路の長さ (`connections`) がある駅間ではそれを走行距離に
使い、無い駅間だけ直線距離 × 迂回係数で見積もり、その距離で求め直した較正
(`speed_table` / `segment_speed_table`) を使います。`estimateArrivalTimes` も同じ
計算です。乗換経路探索 (`connectedRoutes`) の所要時間だけは、元の計算 (直線距離 ×
迂回係数と、`domain/legacy_speed_table.rs` の元の較正) のままです。そのため、
経路検索の所要時間と ETA は一致しません。`Legacy` の値で台形の速度プロファイルを作って走らせると、
`estimateArrivalTimes` より短い時間で走り切ります (#1709)。所要時間を推定に
合わせたいクライアントは、`Estimated` の見込みを使います。

区間の切り出しには両者で同じ関数を使い、環状線では継ぎ目をまたぐ短いほうの
弧を選ぶので、両者の駅の並びは一致します (`lineGroupId` を指定した
`trainRoute` は、従来どおり格納順で切り出します)。次のいずれかに当てはまる
Expand All @@ -487,6 +513,23 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int!
- 両端の駅が `fromStationId` / `toStationId` と一致しない
- `viaLineIds`・`directionId`・`lineGroupId` と同時に指定されている

### 所要時間のベンチマーク (`travel_times/`)

到着時間推定 (`estimateArrivalTimes` と `trainRoute` の `Estimated`) の所要時間を、実際の列車の所要時間と比べる基準を
`travel_times/cases.csv` に置いています。速度の較正テーブルや一般則は、1 つの路線に合わせて変えると、同じ
規則を使うほかの路線の推定も変わります。変更の前後で全体の誤差を測るための仕組み
です。

- `cargo test -p stationapi-worker` (`src/travel_times.rs`) は、基準ごとの「実際の
典型的な所要時間からのずれ」を `travel_times/baseline.csv` の記録と比べ、悪くなる
と失敗します。記録は本番と同じ生成データで作るので、比べるのは `generated/` で
動くときだけです。
CI では `build_worker.yml` が生成データを作ってから走らせます。
- `make travel-time-report` は、生成データで動く Worker に問い合わせて全件の誤差を
出します。推定の規則や較正を変える PR には、変更前と変更後のレポートを載せます。

基準の決め方と記録の更新方法は `travel_times/README.md` にあります。

### 行き先の検索 (`stationsByName`)

`stationsByName` に `fromStationGroupId` を指定すると、その駅から行ける駅だけに
Expand Down Expand Up @@ -724,6 +767,7 @@ repository の実装がないメソッドは、空の結果ではなく `DomainE
│ │ ├── entity/ # Station / Line / TrainType / Company ...
│ │ ├── repository/ # 抽象インターフェース
│ │ ├── arrival_estimation.rs
│ │ ├── legacy_speed_table.rs # 元の較正 (connectedRoutes と trainRoute の Legacy が使う)
│ │ ├── route_search.rs # 乗換経路探索 (RAPTOR)
│ │ ├── route_topology.rs # 所要時間を持たない系統網 (stationsByName の到達判定)
│ │ ├── segment_speed_table.rs
Expand Down
9 changes: 8 additions & 1 deletion schema/public.graphql
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,11 @@ enum ConnectedRouteSort {
TransferCount
}

enum TrainRouteModel {
Legacy
Estimated
}

enum TtsAlphabet {
TtsAlphabetUnspecified
Ipa
Expand Down Expand Up @@ -100,7 +105,7 @@ type Query {
routeTypes(fromStationGroupId: Int!, toStationGroupId: Int!, viaLineId: Int, pageSize: Int, pageToken: String): RouteTypePage!
connectedRoutes(fromStationGroupId: Int!, toStationGroupId: Int!, viaLineId: Int, sortBy: ConnectedRouteSort): [ConnectedRoute!]!
estimateArrivalTimes(fromStationId: Int!, toStationId: Int!, viaLineIds: [Int!], directionId: Int, legs: [RouteLegInput!]): EstimatedArrivalPage!
trainRoute(fromStationId: Int!, toStationId: Int!, lineGroupId: Int, legs: [RouteLegInput!]): TrainRouteResponse!
trainRoute(fromStationId: Int!, toStationId: Int!, lineGroupId: Int, legs: [RouteLegInput!], model: TrainRouteModel): TrainRouteResponse!
}

type ConnectedRoute {
Expand Down Expand Up @@ -330,6 +335,8 @@ type TrainRouteSegment {
maxSpeed: Float
maxAcceleration: Float
maxDeceleration: Float
arrivalCumulativeMinutes: Float
departureCumulativeMinutes: Float
}

type TrainRouteResponse {
Expand Down
9 changes: 8 additions & 1 deletion scripts/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -48,9 +48,16 @@ OSM データは [Open Database License (ODbL)](https://www.openstreetmap.org/co
## compute_speed_table.py

公開 GTFS 時刻表から、到着時間推定(`arrival_estimation.rs`)の速度較正テーブル
2 種類を再計算します。運動学モデルを Python で再現し、実ダイヤの所要時間を
2 種類を再計算します。このテーブルを使うのは `estimateArrivalTimes` と `trainRoute` の
`Estimated` です (`connectedRoutes` と `trainRoute` の `Legacy` は `legacy_speed_table.rs`
の元の較正を使います)。運動学モデルを Python で再現し、実ダイヤの所要時間を
再現する実効最高速度を二分探索でフィッティングします。

駅間の距離は、推定と同じく線路の長さ(`generated/connections.csv`)を使い、無い
駅間だけ直線距離 × 迂回係数で見積もります。先に `make data` で `generated/` を
作ってから実行してください。較正を変えたら、`make travel-time-report` で全体の
誤差が悪くならないことを確かめます(`travel_times/README.md`)。

1. **路線 × 列車種別**(`stationapi/src/domain/speed_table.rs`):
列車全体の所要時間へのフィット。一般則(路線種別の基本速度 × 種別倍率)から
±10% 以上乖離した路線だけを出力します。
Expand Down
50 changes: 41 additions & 9 deletions scripts/compute_speed_table.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,10 @@
TYPES_CSV = os.path.join(ROOT, "data", "4!types.csv")
SST_CSV = os.path.join(ROOT, "data", "5!station_station_types.csv")
SPEED_TABLE_RS = os.path.join(ROOT, "stationapi", "src", "domain", "speed_table.rs")
# 隣り合う駅の線路の長さ。preprocessor (make data) が N02 から求め、Worker に埋め込む
# ものと同じ。到着時間推定はこれがある駅間では迂回係数ではなく線路の長さを使うので、
# 較正も同じ距離で行う。
CONNECTIONS_CSV = os.path.join(ROOT, "generated", "connections.csv")
CACHE_DIR = os.path.join(HERE, ".gtfs_cache")

HTTP_TIMEOUT = 120
Expand All @@ -71,7 +75,7 @@ class Feed:
Feed(
key="hakodate_tram",
name="函館市電",
url="https://api-public.odpt.org/api/v4/files/odpt/HakodateCity/Alllines.zip?date=20260615",
url="https://api-public.odpt.org/api/v4/files/odpt/HakodateCity/Alllines.zip?date=20260815",
needs_token=False,
license="公共交通オープンデータセンター(認証なし公開)",
),
Expand Down Expand Up @@ -309,6 +313,32 @@ def load_repo_data():
return lines, by_line, kinds, groups


def load_track_distances() -> dict[tuple[int, int], float]:
"""隣り合う 2 駅の線路の長さ(m)。キーは (小さい station_cd, 大きい station_cd)。"""
if not os.path.isfile(CONNECTIONS_CSV):
raise SystemExit(
f"{CONNECTIONS_CSV} がありません。make data で generated/ を作ってから実行してください"
"(到着時間推定は線路の長さを使うので、較正も同じ距離で行う)"
)
track: dict[tuple[int, int], float] = {}
with open(CONNECTIONS_CSV, newline="", encoding="utf-8") as f:
for r in csv.DictReader(f):
a, b = int(r["station_cd1"]), int(r["station_cd2"])
d = float(r["distance"])
if d > 0:
track[(min(a, b), max(a, b))] = d
return track


def pair_distance(track: dict, a: dict, b: dict, alpha: float) -> float:
"""駅 a→b の走行距離(m)。estimate_arrival_minutes_with_track と同じく、線路の長さが
あればそれを、無ければ直線距離 × 迂回係数を使う。"""
d = track.get((min(a["cd"], b["cd"]), max(a["cd"], b["cd"])))
if d:
return d
return haversine(a["lat"], a["lon"], b["lat"], b["lon"]) * alpha


def line_detour(lines: dict, by_line: dict, line_cd: int) -> float:
"""路線全体で較正した迂回係数 α(estimate_arrival_minutes_calibrated と同等)。"""
li = lines[line_cd]
Expand Down Expand Up @@ -680,7 +710,7 @@ def segment_sanity_upper_kmh(base: float, line_type: int | None) -> float:


def calibrate_feed(
feed: Feed, zf: zipfile.ZipFile, lines, by_line, kinds, groups
feed: Feed, zf: zipfile.ZipFile, lines, by_line, kinds, groups, track
) -> tuple[list[Calibration], dict[tuple[int, int, int], list[tuple[float, str]]]]:
stops_by_id = {
s["stop_id"]: (norm_name(s["stop_name"]), float(s["stop_lat"]), float(s["stop_lon"]))
Expand Down Expand Up @@ -760,8 +790,7 @@ def calibrate_feed(
alpha = line_detour(lines, by_line, line_cd)
seq: list[tuple[float, bool]] = [(0.0, True)]
for a, b in zip(span, span[1:]):
d = haversine(a["lat"], a["lon"], b["lat"], b["lon"]) * alpha
seq.append((d, b["cd"] in served))
seq.append((pair_distance(track, a, b, alpha), b["cd"] in served))
obs = arr_last - dep0

dedup_key = (tuple(s["cd"] for s in span), tuple(sorted(served)), round(obs, 1))
Expand Down Expand Up @@ -989,6 +1018,7 @@ def main() -> int:

token = os.environ.get("ODPT_ACCESS_TOKEN")
lines, by_line, kinds, groups = load_repo_data()
track = load_track_distances()

all_results: list[Calibration] = []
all_seg_samples: dict[tuple[int, int, int], list[tuple[float, str]]] = {}
Expand All @@ -1005,7 +1035,7 @@ def main() -> int:
if zf is None:
continue
with zf:
results, seg_samples = calibrate_feed(feed, zf, lines, by_line, kinds, groups)
results, seg_samples = calibrate_feed(feed, zf, lines, by_line, kinds, groups, track)
all_results.extend(results)
for key, ts in seg_samples.items():
all_seg_samples.setdefault(key, []).extend(ts)
Expand Down Expand Up @@ -1043,9 +1073,10 @@ def line_baseline(line_cd: int) -> float:
return general_rule_speed(lines[line_cd]["line_type"], 0)

name_of = {s["cd"]: s["name"] for sts in by_line.values() for s in sts}
coord_of = {s["cd"]: (s["lat"], s["lon"]) for sts in by_line.values() for s in sts}
station_of = {s["cd"]: s for sts in by_line.values() for s in sts}

# ペアごとの平均ターゲットとみなし距離。距離には路線較正済みの迂回係数を使う。
# ペアごとの平均ターゲットとみなし距離。距離は線路の長さ (無ければ路線較正済みの
# 迂回係数で見積もった直線距離) を使う。
mean_targets: dict[tuple[int, int, int], float] = {}
pair_dist_m: dict[tuple[int, int, int], float] = {}
# 出発間隔ベースでしか較正できなかったペア(繰り越し補正の対象)。
Expand All @@ -1056,7 +1087,6 @@ def line_baseline(line_cd: int) -> float:
continue
if line_cd not in detour_cache:
detour_cache[line_cd] = line_detour(lines, by_line, line_cd)
(lat1, lon1), (lat2, lon2) = coord_of[cd_lo], coord_of[cd_hi]
key = (line_cd, cd_lo, cd_hi)
# 到着時刻ベース(純走行時間)が十分あればそれだけを使う。無いフィード
# (東京メトロは全駅 arr==dep)では出発間隔ベースの上側トリム平均で
Expand All @@ -1070,7 +1100,9 @@ def line_baseline(line_cd: int) -> float:
used = dep_ts[:keep]
dep_only_keys.add(key)
mean_targets[key] = sum(used) / len(used)
pair_dist_m[key] = haversine(lat1, lon1, lat2, lon2) * detour_cache[line_cd]
pair_dist_m[key] = pair_distance(
track, station_of[cd_lo], station_of[cd_hi], detour_cache[line_cd]
)

raw_targets = dict(mean_targets)
rebalance_line_targets(mean_targets, pair_dist_m, by_line, lines, dep_only_keys)
Expand Down
Loading
Loading