From 6e9e1d5f57d05c8a83bf7f3efbd27c9d000d878d Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Tue, 29 Sep 2026 03:19:51 +0900 Subject: [PATCH 1/2] =?UTF-8?q?=E9=9A=A3=E3=82=8A=E5=90=88=E3=81=86?= =?UTF-8?q?=E9=A7=85=E3=81=AE=E3=81=82=E3=81=84=E3=81=A0=E3=81=AE=E7=B7=9A?= =?UTF-8?q?=E8=B7=AF=E3=81=AE=E9=95=B7=E3=81=95=E3=82=92=E8=BF=94=E3=81=9B?= =?UTF-8?q?=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB=E3=81=97=E3=81=9F=20(#1705?= =?UTF-8?q?)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * 隣り合う駅のあいだの線路の長さを国土数値情報から求め、Station.trackDistanceFromPreviousで返すようにした Co-Authored-By: Claude Opus 5.5 * N02の取得にタイムアウトを明示し、CIでN02をキャッシュするようにした Co-Authored-By: Claude Opus 5.5 --------- Co-authored-by: Claude Opus 5.5 --- .github/actions/build-worker/action.yml | 16 +- .gitignore | 1 + AGENTS.md | 11 +- Makefile | 3 +- README.md | 12 +- build.rs | 67 +- data/README.md | 31 +- data_validator/src/main.rs | 69 +- docs/architecture.md | 94 ++- docs/route-search.md | 2 +- preprocessor/src/emit.rs | 3 + preprocessor/src/main.rs | 12 +- preprocessor/src/rail.rs | 16 +- preprocessor/src/table.rs | 6 + preprocessor/src/track/mod.rs | 589 ++++++++++++++++++ preprocessor/src/track/network.rs | 468 ++++++++++++++ schema/public.graphql | 2 + src/graphql/types.rs | 2 + src/index.rs | 49 ++ src/repository.rs | 10 + stationapi/src/domain/arrival_estimation.rs | 1 + stationapi/src/domain/entity/station.rs | 4 + .../domain/repository/station_repository.rs | 15 + stationapi/src/domain/route_search.rs | 1 + stationapi/src/model.rs | 4 + stationapi/src/use_case/dto/station.rs | 1 + stationapi/src/use_case/interactor/query.rs | 107 +++- 27 files changed, 1545 insertions(+), 51 deletions(-) create mode 100644 preprocessor/src/track/mod.rs create mode 100644 preprocessor/src/track/network.rs diff --git a/.github/actions/build-worker/action.yml b/.github/actions/build-worker/action.yml index 4b2ded6d..d112168d 100644 --- a/.github/actions/build-worker/action.yml +++ b/.github/actions/build-worker/action.yml @@ -47,9 +47,21 @@ runs: target key: worker-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }} + # N02 の取得先は国の配信サーバーで、取得に失敗すると preprocessor が失敗する。 + # サーバーの一時的な不調で検証やデプロイが止まらないよう、展開した GeoJSON を + # キャッシュする。中身は版で決まるので、キーは版だけにする + # (preprocessor/src/track/mod.rs の CACHE_DIR と揃えること)。 + - name: Cache N02 railway data + uses: actions/cache@v4 + with: + path: data/N02-25 + key: n02-25 + # data/*.csv をそのまま Worker へ渡すと本番と挙動が変わる。列車種別を # 持たない路線には各駅停車の系統を補う必要があり (約2,400行)、バス停と - # バス路線は GTFS / ODPT から起こす必要があるため。 + # バス路線は GTFS / ODPT から、駅間の線路の長さは国土数値情報 (N02) から + # 起こす必要があるため。N02 はトークン不要で、取得に失敗すると preprocessor + # 自体が失敗する。 - name: Build generated data shell: bash env: @@ -82,7 +94,7 @@ runs: - name: Verify generated data shell: bash run: | - for t in companies lines stations types station_station_types aliases line_aliases; do + for t in companies lines stations types station_station_types aliases line_aliases connections; do test -s "generated/$t.csv" || { echo "::error::$t.csv が空"; exit 1; } echo "$t: $(python3 -c "import csv,sys;print(sum(1 for _ in csv.reader(open(sys.argv[1]))) - 1)" "generated/$t.csv") rows" done diff --git a/.gitignore b/.gitignore index 7ba676c0..6badc5a7 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,7 @@ data/TokyuBus-ShinagawaCity-GTFS/ data/TokyuBus-MeguroCity-GTFS/ data/TokyuBus-ODPT/ data/KeioBus-GTFS/ +data/N02-25/ scripts/.osm_cache/ scripts/.gtfs_cache/ __pycache__/ diff --git a/AGENTS.md b/AGENTS.md index ab85947d..7f9b94c5 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -5,12 +5,12 @@ This guide explains how automation agents and human contributors should work wit ## Project Layout - `src/` – The Worker itself (`stationapi-worker`, wasm32 only). `lib.rs` holds the endpoints, `index.rs` parses the embedded CSVs into in-memory indexes, `repository.rs` implements the repository traits against those indexes, and `graphql/` holds the async-graphql types and resolvers. `index.rs` also holds the spatial grid used by every coordinate lookup — see **Coordinate lookups** below. - `schema/public.graphql` – The published GraphQL schema. CI diffs the Worker's SDL against this file, so an unintended change fails the build. -- `build.rs` – Stages `generated/*.csv` (falling back to `data/*.csv`) into `OUT_DIR` and pre-converts `station_station_types` into a fixed-width binary. +- `build.rs` – Stages `generated/*.csv` (falling back to `data/*.csv`) into `OUT_DIR` and pre-converts `station_station_types` and `connections` into fixed-width binaries. - `wrangler.jsonc` – Staging and production deployment settings. - `stationapi/src/domain/` – Entity definitions and repository abstractions. `repository/` provides `async_trait`-based interfaces, and `normalize.rs` contains text normalization for search. - `stationapi/src/use_case/` – Application logic. `interactor/query.rs` implements the `QueryUseCase` contract defined in `traits/query.rs`; `dto/` converts entities into `model` types (this is where IPA and TTS segments are built). - `stationapi/src/model.rs` – The values the API returns; the layer between entities and GraphQL types. -- `preprocessor/` – Build-time CLI that assembles `generated/*.csv` from `data/*.csv`, the GTFS feeds, and the Tokyu ODPT JSON. +- `preprocessor/` – Build-time CLI that assembles `generated/*.csv` from `data/*.csv`, the GTFS feeds, the Tokyu ODPT JSON, and the MLIT railway data (N02, for track lengths — see **Track distances** below). - `data/` – Canonical CSV datasets. Files follow the `N!table.csv` naming scheme. Detailed instructions are in `data/README.md`. - `data_validator/` – CLI that verifies cross-file constraints (`cargo run -p data_validator`). - `Makefile` – Convenience targets (`make help` lists them all). @@ -25,10 +25,11 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - `ODPT_ACCESS_TOKEN` – ODPT consumer key used to download authenticated data such as Seibu Bus GTFS, Keio Bus GTFS, Tokyu Bus JSON, and the Tokyu-operated Ota, Shinagawa, and Meguro community bus GTFS feeds. Only used by `preprocessor`. - `DISABLE_BUS_FEATURE` – set to `true` to build rail-only data. - Keep local secrets in `.env.local` (git-ignored) and export them before running `make data`. +- `make data` also downloads the MLIT National Land Numerical Information railway data (N02, about 13 MB, no token) and caches its track GeoJSON under `data/N02-25/` (git-ignored). Unlike a bus feed, a failed download fails `preprocessor` — silently shipping data without track lengths would turn every `trackDistanceFromPrevious` into `null`. ## Running and Deploying - **Local development** - 1. `make data` builds `generated/*.csv` from `data/*.csv`, the GTFS feeds, and the Tokyu ODPT JSON. Feeds already extracted under `data/*-GTFS/` are reused; the ODPT JSON is cached for seven days. + 1. `make data` builds `generated/*.csv` from `data/*.csv`, the GTFS feeds, the Tokyu ODPT JSON, and N02. Feeds already extracted under `data/*-GTFS/` are reused; the ODPT JSON is cached for seven days; N02 is reused while `data/N02-25/` exists. 2. `make build` compiles the Worker, `make dev` serves it on `http://127.0.0.1:8787`. 3. `GET /__ping` answers without touching the data, `GET /__health` reports index sizes, `GET /` serves GraphiQL, and `GET /__schema` returns the SDL. - **Deploying** @@ -53,7 +54,8 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - Column sets live in `preprocessor/src/rail.rs` (`*_COLUMNS`). Update them alongside any CSV column change; `generated/*.csv` must keep the same column order because `src/index.rs` reads it by name and `build.rs` by position. - Columns whose name starts with `#` are notes and are not loaded. - **Through-service junction stations** – When a train type runs through a station where its lines connect, add a `5!station_station_types.csv` row for every line-specific `station_cd` at that station, even when those rows share one `station_g_cd`. The only exception is when the train type explicitly identifies a direction or line-specific operation that excludes one side. Omitting either ID makes the train type selectable from only one line in the app. For example, Hida at Gifu must include both the Takayama Main Line station (`1141601`) and the Tokaido Main Line station (`1150239`). Audit both sides whenever adding or editing a through-service pattern. -- `data_validator` currently verifies that `5!station_station_types.csv` references valid station and type IDs, and that order-sensitive station sequences in `3!stations.csv` stay intact under `ORDER BY e_sort, station_cd` (e.g. the Toei Oedo Line's Tochomae rows, whose misordering silently drops the station from ETA estimation). Extend the validator when new cross-references or order-sensitive spots are introduced and keep the process fail-fast (panic on invalid data). +- **`8!connections.csv` holds hand corrections to track lengths.** `preprocessor` computes the length between every pair of adjacent rail stations from N02; a row here (`station_cd1`, `station_cd2`, `distance` in meters, either direction) overrides the computed value. Use it only where N02's geometry or a station's coordinates make the computed value wrong, and cite the source of the corrected value in the pull request. +- `data_validator` currently verifies that `5!station_station_types.csv` references valid station and type IDs, that `8!connections.csv` references valid stations with non-negative distances and no duplicate pair, and that order-sensitive station sequences in `3!stations.csv` stay intact under `ORDER BY e_sort, station_cd` (e.g. the Toei Oedo Line's Tochomae rows, whose misordering silently drops the station from ETA estimation). Extend the validator when new cross-references or order-sensitive spots are introduced and keep the process fail-fast (panic on invalid data). ## Testing and Quality - **Tests** – `make test` runs the unit tests for every native crate, plus `cargo test -p stationapi-worker`. The Worker only *runs* on Workers, but `src/index.rs` is a pure in-memory data structure that builds and executes natively, so its tests (including the grid-versus-full-scan differential check) run here. They need no external services. @@ -75,6 +77,7 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - **GTFS bus integration** – `preprocessor/src/gtfs/` reads the GTFS feeds into an in-memory representation and then projects them onto the shared `stations` / `lines` / `types` / `station_station_types` tables (`gtfs/integrate.rs`). Only routes, stops, trips, and stop_times are read; calendar, shapes, feed_info, and agencies do not affect the output. Every configured GTFS feed is imported, including Seibu Bus and Keio Bus (both downloaded from ODPT with `ODPT_ACCESS_TOKEN`). Tokyu Bus ordinary-route `BusroutePattern`, `BusstopPole`, and `BusTimetable` JSON are converted into the same representation; pattern IDs become `shape_id` values so route variants remain queryable as bus TrainTypes. The Tokyu-operated Ota, Shinagawa, and Meguro community buses use their official GTFS feeds and matching JSON routes are excluded to prevent duplicates. `ODPT_ACCESS_TOKEN` is required for authenticated sources; without it those feeds are skipped with a warning rather than failing the build. Stops whose Tokyu JSON records omit coordinates remain available to name and route queries but not coordinate searches. `transport_type` (0: rail, 1: bus) on both `stations` and `lines` keeps rail and bus records queryable side by side. GTFS IDs are namespaced per feed before import to avoid cross-operator collisions. `line_cd` (100,000,000+), `station_cd` / `station_g_cd` (200,000,000+), and bus `type_cd` / `line_group_cd` (100,000,000+) are all deterministic fnv1a hashes that stay clear of the rail data ranges. Disable the entire bus pipeline with `DISABLE_BUS_FEATURE=true`. - **Bus stop translations (readings & English)** – GTFS-JP `translations.txt` layouts differ per feed, so `load_translations` (`preprocessor/src/gtfs/parse.rs`) resolves columns by header name (Seibu ships 6 columns without `record_sub_id`; Keio and the Tokyu community feeds ship 7) and indexes each `stop_name` translation under both keys it may use: `record_id` (== the stop_id, Seibu — with the "-NN" pole suffix also mapped to the parent stop_id) and `field_value` (== the Japanese stop_name, Keio / Tokyu community, where `record_id` is left empty). `load_stops` then looks a stop's translation up by stop_id first, then by name. Keying only by `record_id` would silently drop every field_value-keyed feed, leaving `station_name_k` filled with the kanji stop_name and `station_name_r` empty. Readings arriving as half-width katakana (`ニシハチオウジ`, Keio / Tokyu community) are folded to full-width via `romaji::to_fullwidth_katakana()` before storage. - **Bus English-name fallback** – When a feed provides no English (`en`) translation for a stop — e.g. Tokyu Bus ordinary-route JSON, which carries only `dc:title` and `odpt:kana` — `stationapi/src/domain/romaji.rs::romaji_display_name()` derives a modified-Hepburn romanization (with macrons for long vowels, matching the curated rail style: Tōkyō / Kyōto / Shin-Ōsaka) from the kana reading, and the GTFS reader fills `stop_name_r` with it. The fallback never overwrites a real `en` value, and a reading with no convertible kana stays `NULL` rather than emitting a partial transcription. Because `stop_name_r` is the single upstream source that fans out into the `stations` projection, `search_by_name`, and the romanized bus route/headsign names, this supplements every English-facing surface at once. When projecting into `stations`, `station_name_rn` is filled with the plain-ASCII spelling via `romaji::strip_macrons()` (Tōkyō → Tokyo), mirroring the rail dataset's `_r` (macron) / `_rn` (macron-free) column pair. +- **Track distances** – `Station.trackDistanceFromPrevious` is the track length in meters from the station before it in the returned list, for clients that total a ride's distance (TrainLCD/MobileApp's ride log). It is distinct from `trainRoute`'s `distanceFromPrevious`, which stays the straight-line distance the app's running simulation relies on. Only `lineStations`, `lineGroupStations`, and `trainRoute` fill it (`attach_track_distances` in `QueryInteractor`, called once the order is final); the first station, the first station of each `trainRoute` leg, sections without N02 geometry, and every other query return `null`, and clients fall back to the straight line there. `preprocessor/src/track/` builds an undirected graph from N02's `RailroadSection` LineStrings, collects every pair adjacent in a line's `(e_sort, station_cd)` order or a line group's `station_station_types.id` order (with and without closed stations, plus the seam of loop services within 3 km), snaps each station to every track within 500 m, and takes the Dijkstra path minimizing `2 × snap offset + track length` — snapping to the nearest track alone picks another line of the same operator at large stations (Honmachi) and detours through a transfer station. It first measures on the operators of both stations' companies (`OPERATOR_ALIASES` maps companies whose name differs from N02's), then retries on every operator for lines running on another company's track; a path longer than max(3 × straight, straight + 5 km) counts as unmeasured, and a result shorter than the straight line is raised to it. Pairs in one station group are 0. `build.rs` writes the sorted pairs to `connections.bin`, and `index::track_distance` binary-searches it, so no index is built at isolate start. When bumping the N02 edition, change the URL and cache directory in `preprocessor/src/track/mod.rs` and the `Cache N02 railway data` step in `.github/actions/build-worker/action.yml` together and compare the unmeasured count in the `preprocessor` log. `docs/architecture.md` (駅間の線路の長さ) has the details. - **TTS metadata** – `Station`, `StationNested`, `Line`, `LineNested`, `TrainType`, and `TrainTypeNested` expose `name_ipa` / `name_roman_ipa` plus `name_tts_segments` for multi-segment pronunciation output. Use `name_tts_segments` when clients need per-token SSML construction for mixed-language names such as `Kasai-Rinkai Park`. - **Connected routes** – `connectedRoutes` finds transfer routes automatically, like a journey planner, using a frequency-based RAPTOR search in `stationapi/src/domain/route_search.rs`. Each rail line group is a pattern, station groups are the transfer nodes, and ride times come from `arrival_estimation`; bus lines are excluded. The cost adds a per-boarding wait by `TrainTypeKind` (limited express 15 min, express / high-speed rapid 5 min, others 3 min) and a 3-minute transfer walk — without the wait, infrequent limited expresses would beat the Yamanote Line. Rounds give the time/transfer Pareto set; alternatives come from re-searching with one leg's parallel line groups banned along that leg (at most 8 searches), and are dropped beyond 1.15 × best + 15 min or with two more transfers than the Pareto set. Alternative routes that stop at the same station group in two different legs (backtracking to re-board a banned train) are dropped; pass-through stations are not counted, and the Pareto routes of the first search are never dropped this way (otherwise a station `stationsByName` reports as reachable could get no route). Results are ranked by cost + 5 min per transfer and capped at 6. The time and transfer count stay internal (the API does not return them), so ordering happens on the server: `sortBy: ConnectedRouteSort` picks `Recommended` (the ranking above, the default when omitted), `ArrivalTime` (the estimated time `estimateArrivalTimes` also reports — excluding the first train's wait — then fewer transfers), or `TransferCount` (fewer transfers, then earlier arrival). `route_search::sort_journeys` only reorders the set `search` returned, with a stable sort so ties keep the recommended order; the set itself never depends on `sortBy`. The network (every rail line group plus its time estimates, about 190 ms natively) is built lazily into a `OnceLock` by `StationRepository::get_route_network` on the first `connectedRoutes` call, so other queries never pay for it. Each route is a list of `legs` shaped for the app's one-train-at-a-time flow: every leg carries its boarding and alighting `Station`, both on the line of the line group the search rode — so at a transfer the previous leg's alighting station and the next leg's boarding station may be different stations of one station group — `stationGroupIds`, the station groups from boarding to alighting in travel order including pass-through stations (the search's `JourneyLeg.station_group_ids` as-is — station groups rather than station IDs so the client can match them against whichever train type it picks, which may run on another line; a group appears twice on patterns such as the Oedo Line's Tochomae), and `trainTypes`, every train type usable on that leg (real `groupId`s, so the client picks one and calls `lineGroupStations`): it is exactly `routeTypes(boarding station group, alighting station group, alighting station's line)` (same dedup, same `lines`, same order — the use case calls `get_train_types`), because the search collapses parallel services such as local and rapid into one route and the app needs them to list types and default to the local. `viaLineId`, like `routeTypes`, is the line of the tapped search result and keeps only routes whose last leg arrives on that line. `estimateArrivalTimes` and `trainRoute` accept `legs: [RouteLegInput!]` (the `groupId` of the train type picked from each leg's `trainTypes`, plus the leg's `fromStation.id` and `toStation.id`) and then return values for the whole transfer route. A leg endpoint missing from the chosen line group is matched by station group (the picked local may stop at another line's station of the same group), preferring an exact `station_cd` and, among same-group candidates, the pair giving the shortest slice (through services list two stations of a junction group). ETA estimates each leg on its own line group only and chains them from the origin, adding the 3-minute walk and the next train type's wait at each transfer (the same allowance the search ranks by), returning one route with an empty `id`; `trainRoute` concatenates each leg's segments (each leg restarts at distance 0). Both slice legs with the same function, taking the shorter arc on loop lines, so their station sequences match. More than `MAX_RIDES` (6) legs — more than `connectedRoutes` ever returns — legs that do not connect, ends that differ from `fromStationId` / `toStationId`, or combining `legs` with `viaLineIds` / `directionId` / `lineGroupId` are errors. `docs/architecture.md` (乗換経路探索) has the details, and `docs/route-search.md` documents the search internals (data structures, the scan, pruning, alternatives, determinism). - Changes to the published contract require coordinated updates to `schema/public.graphql`, the async-graphql types in `src/graphql/`, and, when the shape of a value changes, `stationapi/src/model.rs` and the DTO conversions. diff --git a/Makefile b/Makefile index fae6f57d..779fd12b 100644 --- a/Makefile +++ b/Makefile @@ -14,7 +14,7 @@ help: @echo " check - Type-check every crate (worker targets wasm32)" @echo " fmt - Check formatting" @echo " clippy - Lint every crate" - @echo " data - Rebuild generated/*.csv from data/ and the GTFS feeds" + @echo " data - Rebuild generated/*.csv from data/, the GTFS feeds, and the MLIT railway data" @echo " build - Build the Worker (wasm)" @echo " dev - Run the Worker locally (wrangler dev)" @echo " deploy - Deploy to staging (dev branch only)" @@ -46,6 +46,7 @@ clippy: cargo clippy --target wasm32-unknown-unknown -p stationapi-worker --all-targets -- -D warnings # Worker が読むデータを作り直す。data/*.csv や GTFS が変わったら実行する。 +# 駅間の線路の長さに使う国土数値情報 (N02) は data/N02-25/ にキャッシュする。 data: cargo run --profile tool -p stationapi-preprocessor diff --git a/README.md b/README.md index 8c93e42e..12a26397 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,16 @@ A GraphQL API that provides nearby Japanese train stations and bus stops, runnin This project includes a comprehensive dataset of Japanese railway information in the `data/` directory. The data is maintained in CSV format and contributions are primarily targeted at Japanese speakers. For detailed information about data structure and contribution guidelines, please refer to [data/README.md](data/README.md). +## Data Sources + +- Track lengths between adjacent stations (`Station.trackDistanceFromPrevious`) are + derived from the MLIT National Land Numerical Information railway data (N02), + licensed under CC BY 4.0: + 「国土数値情報(鉄道データ)」(国土交通省) + (https://nlftp.mlit.go.jp/ksj/gml/datalist/KsjTmplt-N02-2025.html) を加工して作成 +- Bus stops and routes are derived from the GTFS and ODPT feeds listed in + `preprocessor/src/gtfs/feed.rs` and `preprocessor/src/gtfs/odpt.rs`. + ## Contributors ✨ Thanks goes to these wonderful people ([emoji key](https://allcontributors.org/docs/en/emoji-key)): @@ -62,7 +72,7 @@ into the WASM binary at build time. rustup target add wasm32-unknown-unknown cargo install worker-build --locked -make data # build generated/*.csv from data/ and the GTFS feeds +make data # build generated/*.csv from data/, the GTFS feeds, and the MLIT railway data make build # build the Worker (wasm) make dev # run it locally on http://127.0.0.1:8787 ``` diff --git a/build.rs b/build.rs index 95b8c3dd..4fb2a683 100644 --- a/build.rs +++ b/build.rs @@ -1,8 +1,11 @@ -//! station_station_types.csv を固定長バイナリへ事前変換する。 +//! station_station_types.csv と connections.csv を固定長バイナリへ事前変換する。 //! -//! この CSV は 41,250 行あり、isolate 起動時の CSV パースがコールドスタートの -//! 大半を占める。全列が整数なので 1 行 = i32 x 4 の固定長にしておけば、 -//! ランタイムではスライスを読むだけで済む。 +//! station_station_types は 41,250 行あり、isolate 起動時の CSV パースが +//! コールドスタートの大半を占める。全列が整数なので 1 行 = i32 x 4 の固定長に +//! しておけば、ランタイムではスライスを読むだけで済む。 +//! +//! connections (隣り合う駅のあいだの線路の長さ) は駅の組で並べた 1 行 = i32 x 3 に +//! しておき、ランタイムは索引を作らずに二分探索で引く。 use std::{env, fs, path::Path, path::PathBuf}; @@ -62,6 +65,7 @@ fn main() { stage_csv(&out_dir, "types.csv", "data/4!types.csv"), stage_csv(&out_dir, "aliases.csv", "data/6!aliases.csv"), stage_csv(&out_dir, "line_aliases.csv", "data/7!line_aliases.csv"), + stage_csv(&out_dir, "connections.csv", "data/8!connections.csv"), // station_station_types は下の sst 変換でも参照するが、 // 混在判定に含めるためここでも存在を見る Path::new("generated/station_station_types.csv").is_file(), @@ -140,4 +144,59 @@ fn main() { fs::write(out_dir.join("sst.bin"), &out).expect("sst.bin を書けない"); println!("cargo:warning=sst.bin: {} 行", out.len() / 16); + + write_connections(&out_dir); +} + +/// connections.csv を (station_cd1, station_cd2, 整数メートル) の固定長バイナリへ +/// 変換する。組は小さい station_cd を先にして昇順に並べる (ランタイムの二分探索用)。 +/// +/// data/8!connections.csv へフォールバックした場合は手入力の行だけになる。 +/// 並びや値の書式が生成物と違ってもよいよう、ここで正規化する。 +fn write_connections(out_dir: &Path) { + let mut reader = csv::ReaderBuilder::new() + .has_headers(true) + .from_path(out_dir.join("connections.csv")) + .expect("connections.csv を開けない"); + let headers = reader.headers().expect("ヘッダを読めない").clone(); + let col = |name: &str| { + headers + .iter() + .position(|h| h.trim() == name) + .unwrap_or_else(|| panic!("connections.csv に {name} 列が無い")) + }; + let (i_a, i_b, i_distance) = (col("station_cd1"), col("station_cd2"), col("distance")); + + let mut rows: Vec<(i32, i32, i32)> = Vec::new(); + for record in reader.records() { + let record = record.expect("connections.csv の行を読めない"); + let parse = |i: usize| record.get(i).map(str::trim).unwrap_or(""); + let (Ok(a), Ok(b)) = (parse(i_a).parse::(), parse(i_b).parse::()) else { + panic!("connections.csv の駅コードが整数ではない: {record:?}"); + }; + let distance = parse(i_distance) + .parse::() + .ok() + .filter(|d| d.is_finite() && *d >= 0.0 && *d <= i32::MAX as f64) + .unwrap_or_else(|| panic!("connections.csv の distance が不正: {record:?}")); + rows.push((a.min(b), a.max(b), distance.round() as i32)); + } + rows.sort_unstable(); + for w in rows.windows(2) { + assert!( + (w[0].0, w[0].1) != (w[1].0, w[1].1), + "connections.csv に同じ駅の組が 2 行ある: {} - {}", + w[0].0, + w[0].1 + ); + } + + let mut out: Vec = Vec::with_capacity(rows.len() * 12); + for (a, b, distance) in &rows { + for value in [*a, *b, *distance] { + out.extend_from_slice(&value.to_le_bytes()); + } + } + fs::write(out_dir.join("connections.bin"), &out).expect("connections.bin を書けない"); + println!("cargo:warning=connections.bin: {} 行", rows.len()); } diff --git a/data/README.md b/data/README.md index 6653078e..a2dd142c 100644 --- a/data/README.md +++ b/data/README.md @@ -13,7 +13,7 @@ | `5!station_station_types.csv` | 駅と列車種別の関連情報 | | `6!aliases.csv` | 路線の別名・愛称情報 | | `7!line_aliases.csv` | 駅と路線別名の関連情報 | -| `8!connections.csv` | 駅間の接続・距離情報 | +| `8!connections.csv` | 駅間の線路の長さの手修正 | ## 🏢 1!companies.csv - 鉄道会社情報 @@ -199,26 +199,31 @@ - `station_cd`は`3!stations.csv`に存在する値を使用 - `alias_cd`は`6!aliases.csv`に存在する値を使用 -## 🚇 8!connections.csv - 駅間の接続・距離情報 +## 🚇 8!connections.csv - 駅間の線路の長さの手修正 -> ⚠️ **注意**: このファイルは現在どこでも使用されていません。将来的に経路計算機能で使用される予定ですが、実装時期は未定です。 +隣り合う駅のあいだの線路の長さ(`Station.trackDistanceFromPrevious`)は、 +preprocessor が国土数値情報の鉄道データ(N02)から自動で求めます。このファイルは、 +その値が実際と合わない区間を手で直すためのものです。ここに書いた値は計算結果より +優先されます。計算の方法は [docs/architecture.md の「駅間の線路の長さ」](../docs/architecture.md#駅間の線路の長さ) +を参照してください。 ### フィールド説明 -| フィールド名 | 型 | 必須 | 説明 | 例 | -| ------------- | ---- | ---- | -------------------- | ------------- | -| `id` | 数値 | ✓ | 主キー(設計未確定) | `1` | -| `station_cd1` | 数値 | ✓ | 起点駅コード | `100201` | -| `station_cd2` | 数値 | ✓ | 終点駅コード | `100202` | -| `distance` | 数値 | - | 駅間距離(メートル) | `6140.152858` | +| フィールド名 | 型 | 必須 | 説明 | 例 | +| ------------- | ---- | ---- | ------------------------------ | --------- | +| `id` | 数値 | ✓ | 行番号(`DEFAULT` でもよい) | `1` | +| `station_cd1` | 数値 | ✓ | 駅コード | `100201` | +| `station_cd2` | 数値 | ✓ | 駅コード(`station_cd1` の隣) | `100202` | +| `distance` | 数値 | ✓ | 線路の長さ(メートル) | `6800` | ### 入力時の注意点 - 駅コードは`3!stations.csv`に存在する値を使用 -- 距離はメートル単位で入力 -- 方向性がある場合は、両方向のレコードを作成 -- **現在は使用されていないため、データ入力の優先度は低い** -- **将来的な実装時に仕様が変更される可能性がある** +- 2 駅は、`lineStations` / `lineGroupStations` が返す並びで隣り合う駅にする(それ以外の組は API で使われない) +- 組の向きは問わない。同じ組(向きを入れ替えたものを含む)を 2 行書かない +- 距離はメートル単位の 0 以上の数値。小数は四捨五入して整数メートルになる +- 値の出典(営業キロ・実キロなど)は PR に書く +- `cargo run -p data_validator` が上記を検査する ## 📝 共通ガイドライン diff --git a/data_validator/src/main.rs b/data_validator/src/main.rs index dec342ea..bbcc0f8e 100644 --- a/data_validator/src/main.rs +++ b/data_validator/src/main.rs @@ -123,10 +123,18 @@ fn main() -> Result<(), Box> { println!("[INVALID] {message}"); } + let mut rdr = ReaderBuilder::new().from_path(data_path.join("8!connections.csv"))?; + let connection_records: Vec = rdr.records().collect::, _>>()?; + let invalid_connections = validate_connections(&connection_records, &station_ids); + for message in &invalid_connections { + println!("[INVALID] {message}"); + } + let has_err = !invalid_station_ids.is_empty() || !invalid_type_ids.is_empty() || !invalid_line_ids.is_empty() - || !invalid_station_orders.is_empty(); + || !invalid_station_orders.is_empty() + || !invalid_connections.is_empty(); if has_err { let report = build_markdown_report( @@ -134,6 +142,7 @@ fn main() -> Result<(), Box> { &invalid_type_ids, &invalid_line_ids, &invalid_station_orders, + &invalid_connections, ); let report_path = std::env::var("VALIDATION_REPORT_PATH").unwrap_or("/tmp/validation_report.md".into()); @@ -194,11 +203,57 @@ fn validate_station_orders(station_records: &[StringRecord]) -> Vec { errors } +/// `8!connections.csv` (手で直した駅間の線路の長さ) の各行について、両駅が +/// `3!stations.csv` にあり、別の駅で、長さが 0 以上の数値で、同じ組 +/// (向きを問わない) が 2 行無いことを検証する。preprocessor はこの値を +/// 国土数値情報から求めた値より優先する。 +fn validate_connections(records: &[StringRecord], station_ids: &HashSet) -> Vec { + const COL_STATION_CD1: usize = 1; + const COL_STATION_CD2: usize = 2; + const COL_DISTANCE: usize = 3; + + let mut errors: Vec = Vec::new(); + let mut seen: HashSet<(u32, u32)> = HashSet::new(); + for record in records { + let line = record.iter().collect::>().join(","); + let station = |i: usize| record.get(i).and_then(|v| v.trim().parse::().ok()); + let (Some(a), Some(b)) = (station(COL_STATION_CD1), station(COL_STATION_CD2)) else { + errors.push(format!("8!connections.csv: 駅コードを読めません: {line}")); + continue; + }; + for cd in [a, b] { + if !station_ids.contains(&cd) { + errors.push(format!( + "8!connections.csv: 存在しない station_cd {cd} を参照しています: {line}" + )); + } + } + if a == b { + errors.push(format!("8!connections.csv: 同じ駅どうしの行です: {line}")); + } + let distance = record + .get(COL_DISTANCE) + .and_then(|v| v.trim().parse::().ok()); + if !distance.is_some_and(|d| d.is_finite() && d >= 0.0) { + errors.push(format!( + "8!connections.csv: distance が 0 以上の数値ではありません: {line}" + )); + } + if !seen.insert((a.min(b), a.max(b))) { + errors.push(format!( + "8!connections.csv: 同じ駅の組 (向きを問わない) が 2 行あります: {line}" + )); + } + } + errors +} + fn build_markdown_report( invalid_station_ids: &[String], invalid_type_ids: &[String], invalid_line_ids: &[String], invalid_station_orders: &[String], + invalid_connections: &[String], ) -> String { let mut md = String::new(); @@ -273,6 +328,18 @@ fn build_markdown_report( md.push('\n'); } + if !invalid_connections.is_empty() { + md.push_str(&format!( + "### 駅間の線路の長さのエラー ({} 件)\n\n", + invalid_connections.len() + )); + md.push_str("`8!connections.csv` の行が不正です。\n\n"); + for message in invalid_connections { + md.push_str(&format!("- {}\n", escape_markdown_cell(message))); + } + md.push('\n'); + } + md } diff --git a/docs/architecture.md b/docs/architecture.md index d3b2933c..5f2f56dc 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1,6 +1,6 @@ # StationAPI アーキテクチャドキュメント -> 最終更新: 2026年9月24日 +> 最終更新: 2026年9月29日 ## 目次 @@ -48,21 +48,22 @@ ## 全体構成 ```txt - data/*.csv GTFS (ZIP) ODPT (JSON) - 鉄道の正本データ バス 6 フィード 東急バス - │ │ │ - └────────────────────┴─────────────────┘ + data/*.csv GTFS (ZIP) ODPT (JSON) N02 (GeoJSON) + 鉄道の正本データ バス 6 フィード 東急バス 国土数値情報の線路 + │ │ │ │ + └────────────────────┴─────────────────┴─────────────────┘ │ ┌─────────▼──────────┐ │ preprocessor │ Rust のみで実装。各駅停車の - │ (ビルド時ツール) │ 系統生成と GTFS の統合を行う + │ (ビルド時ツール) │ 系統生成、駅間の線路の長さの + │ │ 計算、GTFS の統合を行う └─────────┬──────────┘ │ - generated/*.csv 7 テーブル + generated/*.csv 8 テーブル │ ┌─────────▼──────────┐ - │ build.rs │ CSV を OUT_DIR に配置し、 - │ │ sst を固定長バイナリに変換 + │ build.rs │ CSV を OUT_DIR に配置し、sst と + │ │ connections を固定長バイナリに変換 └─────────┬──────────┘ │ ┌─────────▼──────────┐ @@ -148,6 +149,8 @@ UseCase 層からはデータベースを使っていた頃と同じインター いません - バス停・バス路線・バス系統は、GTFS と ODPT の JSON から生成する必要が あります +- 隣り合う駅のあいだの線路の長さは、国土数値情報の鉄道データから求める必要が + あります これらを担うのが `preprocessor` crate です。 @@ -159,11 +162,13 @@ make data # cargo run --profile tool -p stationapi-preprocessor 1. `data/*.csv` を読み込む (`#` で始まる列は読み込まない) 2. 各駅停車の系統を生成する (`generate_virtual_local_rail_services`) -3. GTFS フィード 6 本を取得・展開して読み込む (都営バス・西武バス・京王バスと、 +3. 国土数値情報の鉄道データ (N02) を取得し、隣り合う駅のあいだの線路の長さを + 求める (`track::generate_connections`。[駅間の線路の長さ](#駅間の線路の長さ)) +4. GTFS フィード 6 本を取得・展開して読み込む (都営バス・西武バス・京王バスと、 東急バスが運行する大田区・品川区・目黒区のコミュニティバス) -4. 東急バスの ODPT JSON を読み込む (7 日間キャッシュする) -5. バスのデータを lines / stations / types / station_station_types に統合する -6. `generated/*.csv` に書き出す (7 テーブル) +5. 東急バスの ODPT JSON を読み込む (7 日間キャッシュする) +6. バスのデータを lines / stations / types / station_station_types に統合する +7. `generated/*.csv` に書き出す (8 テーブル) 取得や読み込みに失敗したフィードは、警告を出して飛ばします。ただし、GTFS の 路線を 1 つも取り込めなかった場合は、バスが丸ごと欠けたデータを出さないよう @@ -173,6 +178,59 @@ make data # cargo run --profile tool -p stationapi-preprocessor `station_station_types.id` はそのまま停車順として使われるため、行の順序に 意味があります。書き出すときは必ず `id` の昇順に並べます。 +### 駅間の線路の長さ + +`Station.trackDistanceFromPrevious` は、返す駅の並びで直前にある駅からの +線路の長さ (メートル) です。`trainRoute` の `distanceFromPrevious` は駅の座標 +どうしの直線距離なので、カーブの多い区間では実際より短く出ます。こちらはその +代わりに、乗車距離を集計するクライアントが使います。 + +値は、国土数値情報の鉄道データ (N02、国土交通省、CC BY 4.0) の線路区間 +(`RailroadSection`) から preprocessor で求め、`generated/connections.csv` +(`station_cd1 < station_cd2`、整数メートル) に書き出します。 + +1. **グラフ**: 線路区間の LineString を、頂点を座標 (小数第 5 位) で同一視した + 無向グラフにする。N02-25 で頂点 約 38 万、辺 約 38 万 +2. **駅の組**: API が返す駅の並びで隣り合う鉄道駅の組を集める。並びは路線の + `(e_sort, station_cd)` 順と系統の `station_station_types.id` 順で、廃止駅を + 除いた並びと除かない並びの両方を使う。環状運転の系統は、末尾駅と先頭駅が + 3km 以内なら継ぎ目の組も含める (`trainRoute` は継ぎ目をまたいで切り出す + ため)。同じ駅グループの組 (直通運転の境界駅) は 0 にする +3. **駅を線路へ寄せる**: 駅から 500m 以内の辺をすべて候補にし、 + 「寄せた距離 × 2 + 線路上の距離」が最小になる組を Dijkstra で探す。最も + 近い線路だけに寄せると、大きな駅で同じ事業者の別路線 (本町の御堂筋線と + 四つ橋線など) に寄って乗換駅をまわる遠回りが出るため +4. **事業者で絞る**: まず両駅の路線の会社の線路だけで測る。N02 の事業者名と + 会社名 (`company_name_h` から「株式会社」を除いたもの) が違う会社は + `OPERATOR_ALIASES` で対応させる。他社の線路を走る路線 (北陸新幹線の + 上越妙高以西、相鉄・JR直通線など) はそれでは寄せられないので、全事業者の + 線路で測り直す +5. **打ち切りと下限**: 線路上の距離が直線距離の 3 倍と直線距離 + 5km の大きい方を + 超えたら、線形が途切れて大回りしたものとみなして測れなかったことにする + (実在の最大は木次線の出雲坂根〜三井野原の約 3.6 倍)。線路の長さは直線距離より + 短くならないので、駅の座標のずれで短く出た値は直線距離で抑える + +2026 年 9 月時点のデータでは、約 11,000 組のうち約 10,400 組を計算でき、 +測れなかった約 290 組のほとんどは N02 に線路が残っていない廃線の駅でした +(営業中の駅どうしは 2 組)。実キロと比べると、東海道・山陽新幹線の東京〜博多が +1,066km (実キロ 1,069km)、石勝線のトマム〜新得が 33.8km (営業キロ 33.8km) です。 +preprocessor の実行は 1 秒ほどです。 + +`data/8!connections.csv` に書いた行は計算結果より優先します。N02 の線形や駅の +座標のずれで値が実際と合わない区間は、ここで直します。 + +Worker では `build.rs` が 1 行 = `i32` 3 つの固定長バイナリ (`connections.bin`、 +約 130KB) に変換し、組の昇順に並べます。ランタイムは索引を作らずに二分探索で +引くので、コールドスタートの費用は増えません。値を埋めるのは、並びを返す +`lineStations`・`lineGroupStations`・`trainRoute` だけです (他の問い合わせでは +直前の駅が無いので `null`)。`trainRoute` を `legs` で呼んだ場合は、各区間の +先頭の駅も `null` になります。 + +N02 は `data/N02-25/` にキャッシュします (git 管理外。CI では `actions/cache` で +保存します)。取得できなかった場合、preprocessor は失敗します。版を上げるときは +`preprocessor/src/track/mod.rs` の URL とキャッシュ先、`.github/actions/build-worker/action.yml` +のキャッシュを揃えて変え、測れなかった組の件数を確かめてください。 + ### バスのコード生成 バス由来のレコードには、FNV-1a ハッシュを使って、鉄道と重ならない値域の @@ -210,6 +268,7 @@ PostgreSQL のクエリは、次のように置き換えています。 | `point(lat,lon) <-> point()` | グリッド索引 (`Grid`) で探索半径の内側だけを調べ、haversine で距離を計算する | | `pg_trgm` の GIN インデックス | 全件走査と `contains()` | | `station_station_types` の JOIN | `HashMap` による索引 | +| `connections` の参照 | 並べ済みの固定長バイナリを二分探索 ([駅間の線路の長さ](#駅間の線路の長さ)) | ### 座標による検索 @@ -474,8 +533,8 @@ input RouteLegInput { lineGroupId: Int! fromStationId: Int! toStationId: Int! | 大宮 → 新大阪 | 349ms | 31〜36ms | | 仙台 → 博多 | 693ms | 14〜20ms | -時刻表、運転間隔、駅グループをまたぐ徒歩連絡のデータがない -(`8!connections.csv` は空) ため、待ち時間は種別から見積もった値です。 +時刻表、運転間隔、駅グループをまたぐ徒歩連絡のデータがないため、待ち時間は +種別から見積もった値です。 季節運行の臨時列車も、通常の系統と同じように扱います。 --- @@ -642,7 +701,7 @@ repository の実装がないメソッドは、空の結果ではなく `DomainE . ├── Cargo.toml # stationapi-worker (wasm32 専用) + workspace ├── wrangler.jsonc # staging / production の設定 -├── build.rs # CSV を OUT_DIR に配置し、sst.bin を生成 +├── build.rs # CSV を OUT_DIR に配置し、sst.bin と connections.bin を生成 ├── src/ # Worker 本体 │ ├── lib.rs # エンドポイント │ ├── index.rs # 埋め込みデータのパースと索引 @@ -679,12 +738,13 @@ repository の実装がないメソッドは、空の結果ではなく `DomainE │ └── src/ │ ├── rail.rs # data/*.csv の読み込みと各駅停車の系統生成 │ ├── gtfs/ # GTFS / ODPT の取得・解析・統合 +│ ├── track/ # 国土数値情報からの駅間の線路の長さ │ ├── codes.rs # バス用コードの生成 │ ├── table.rs # 出力テーブルの表現 │ └── emit.rs # CSV の書き出し │ ├── data_validator/ # data/*.csv の整合性チェック -├── data/ # 鉄道の正本データ (CSV) と GTFS の展開先 +├── data/ # 鉄道の正本データ (CSV) と GTFS・N02 の展開先 ├── generated/ # preprocessor の出力 (git 管理外) ├── scripts/ # データ整備とスキーマ比較のスクリプト └── tools/ # IPA カバレッジの監査 diff --git a/docs/route-search.md b/docs/route-search.md index 407607a7..ed7f2272 100644 --- a/docs/route-search.md +++ b/docs/route-search.md @@ -369,7 +369,7 @@ while queue から禁止集合を取り出せる間: 考慮しません。待ち時間は種別から見積もった値です。 - **所要時間は上下で同じとみなす**: 後ろ向きの走査は、上りと下りの所要時間が 同じだと仮定しています。 -- **駅グループをまたぐ徒歩連絡はない**: `8!connections.csv` が空なので、 +- **駅グループをまたぐ徒歩連絡はない**: 徒歩連絡のデータを持たないので、 別の駅グループへ歩いて乗り換える経路は扱いません。 - **乗換の徒歩時間は一律 3 分**: 駅ごとの乗換時間の違いは反映しません。 - **バスは対象外**: 網に含めるのは鉄道の系統だけです。 diff --git a/preprocessor/src/emit.rs b/preprocessor/src/emit.rs index 9cbd052d..07bed9ba 100644 --- a/preprocessor/src/emit.rs +++ b/preprocessor/src/emit.rs @@ -22,6 +22,8 @@ const OUTPUTS: &[(&str, &str)] = &[ ("station_station_types", "id"), ("aliases", "id"), ("line_aliases", "id"), + // track::generate_connections が (station_cd1, station_cd2) 順に採番している。 + ("connections", "id"), ]; pub fn write_all(dataset: &mut Dataset, out_dir: &Path) -> Result<()> { @@ -37,6 +39,7 @@ pub fn write_all(dataset: &mut Dataset, out_dir: &Path) -> Result<()> { "station_station_types" => &mut dataset.sst, "aliases" => &mut dataset.aliases, "line_aliases" => &mut dataset.line_aliases, + "connections" => &mut dataset.connections, other => unreachable!("未知のテーブル {other}"), }; table.sort_by_int_col(order_by); diff --git a/preprocessor/src/main.rs b/preprocessor/src/main.rs index 07eef25b..8340b315 100644 --- a/preprocessor/src/main.rs +++ b/preprocessor/src/main.rs @@ -6,10 +6,14 @@ //! //! ```text //! data/*.csv ─┐ -//! GTFS zip ─┼─> preprocessor ─> generated/*.csv ─> worker-build ─> WASM -//! ODPT JSON ─┘ +//! GTFS zip ─┤ +//! ODPT JSON ─┼─> preprocessor ─> generated/*.csv ─> worker-build ─> WASM +//! N02 GeoJSON ┘ //! ``` //! +//! 隣り合う駅のあいだの線路の長さ (`connections.csv`) は、国土数値情報の +//! 鉄道データ (N02) の線路区間から求める。 +//! //! 使い方: //! //! ```text @@ -25,6 +29,7 @@ mod emit; mod gtfs; mod rail; mod table; +mod track; use std::path::{Path, PathBuf}; @@ -58,6 +63,9 @@ fn main() -> Result<()> { let mut dataset = rail::Dataset::load(data_dir)?; dataset.generate_virtual_local_rail_services()?; + // 生成した各駅停車の系統の並びも使うので、その後に置く。バス停は測らない。 + let network = track::load()?; + track::generate_connections(&mut dataset, &network)?; if bus_feature_disabled() { info!("DISABLE_BUS_FEATURE が立っているのでバスを取り込まない"); diff --git a/preprocessor/src/rail.rs b/preprocessor/src/rail.rs index 2de667c5..d8a3da18 100644 --- a/preprocessor/src/rail.rs +++ b/preprocessor/src/rail.rs @@ -8,7 +8,7 @@ use anyhow::{bail, Context, Result}; use crate::table::{cell_i32, int, Table}; use crate::{info, warn}; -/// 出力する 7 テーブルの列。Worker 側 (`src/index.rs` と `build.rs`) が +/// 出力する 8 テーブルの列。Worker 側 (`src/index.rs` と `build.rs`) が /// この並びを前提に読むので、順序を変えない。 pub const COMPANY_COLUMNS: &[&str] = &[ "company_cd", @@ -111,6 +111,11 @@ pub const ALIAS_COLUMNS: &[&str] = &[ pub const LINE_ALIAS_COLUMNS: &[&str] = &["id", "station_cd", "alias_cd"]; +/// 隣り合う駅のあいだの線路の長さ。`station_cd1 < station_cd2`、`distance` は +/// 整数メートル。`data/8!connections.csv` の行 (手で直した値) に、国土数値情報から +/// 求めた値を足して書き出す (`track::generate_connections`)。 +pub const CONNECTION_COLUMNS: &[&str] = &["id", "station_cd1", "station_cd2", "distance"]; + /// 種別を持たない路線へ補う各駅停車の既定種別。 const DEFAULT_RAIL_TYPE_CD: i32 = 100; /// 「各駅停車」と呼ぶ路線に使う種別。 @@ -144,6 +149,7 @@ pub struct Dataset { pub sst: Table, pub aliases: Table, pub line_aliases: Table, + pub connections: Table, } impl Dataset { @@ -160,6 +166,7 @@ impl Dataset { sst: Table::new(SST_COLUMNS, None), aliases: Table::new(ALIAS_COLUMNS, Some("id")), line_aliases: Table::new(LINE_ALIAS_COLUMNS, Some("id")), + connections: Table::new(CONNECTION_COLUMNS, None), }; load_csv(&mut dataset.companies, &data_dir.join("1!companies.csv"))?; @@ -175,6 +182,10 @@ impl Dataset { &mut dataset.line_aliases, &data_dir.join("7!line_aliases.csv"), )?; + load_csv( + &mut dataset.connections, + &data_dir.join("8!connections.csv"), + )?; // transport_type は CSV に無いので既定値を入れる (0 = 鉄道)。 fill_default(&mut dataset.lines, "transport_type", "0"); @@ -187,7 +198,7 @@ impl Dataset { assign_serial(&mut dataset.sst, "id"); info!( - "取り込み: companies={} lines={} stations={} types={} sst={} aliases={} line_aliases={}", + "取り込み: companies={} lines={} stations={} types={} sst={} aliases={} line_aliases={} connections={}", dataset.companies.len(), dataset.lines.len(), dataset.stations.len(), @@ -195,6 +206,7 @@ impl Dataset { dataset.sst.len(), dataset.aliases.len(), dataset.line_aliases.len(), + dataset.connections.len(), ); Ok(dataset) diff --git a/preprocessor/src/table.rs b/preprocessor/src/table.rs index 7ec432d0..714050d1 100644 --- a/preprocessor/src/table.rs +++ b/preprocessor/src/table.rs @@ -87,6 +87,12 @@ impl Table { true } + /// 全行を捨てる。列の定義は残す。 + pub fn clear(&mut self) { + self.rows.clear(); + self.by_pk.clear(); + } + /// 指定列で行を並べ替える。数値列として比較する。 pub fn sort_by_int_col(&mut self, name: &str) { let idx = self.col(name); diff --git a/preprocessor/src/track/mod.rs b/preprocessor/src/track/mod.rs new file mode 100644 index 00000000..5e1c8237 --- /dev/null +++ b/preprocessor/src/track/mod.rs @@ -0,0 +1,589 @@ +//! 隣り合う駅のあいだの線路の長さ (`connections`)。 +//! +//! 国土数値情報の鉄道データ (N02、国土交通省、CC BY 4.0) の線路区間から求める。 +//! 駅の並び (路線の `e_sort` 順と、系統の `station_station_types.id` 順) で +//! 隣り合う鉄道駅の組すべてについて、2 駅を線路へ寄せ、その間の最短経路の長さを +//! 書き出す。Worker はこれを `Station.trackDistanceFromPrevious` として返す。 +//! +//! `data/8!connections.csv` に書いた値は計算結果より優先する。N02 の線形が +//! 実際と違う区間を手で直すため。 + +mod network; + +use std::collections::{HashMap, HashSet}; +use std::fs::{self, File}; +use std::io::{BufReader, Cursor, Read}; +use std::path::Path; +use std::time::Duration; + +use anyhow::{bail, Context, Result}; +use serde::Deserialize; +use stationapi::domain::arrival_estimation::haversine_distance; +use zip::ZipArchive; + +use crate::rail::Dataset; +use crate::table::{cell_i32, int, text}; +use crate::{info, warn}; +use network::{RailNetwork, Section}; + +/// 国土数値情報 鉄道データ (令和 7 年度)。版を上げるときは [`CACHE_DIR`] と +/// [`GEOJSON_NAME`]、CI のキャッシュ (`.github/actions/build-worker/action.yml`) も +/// 揃えて変える。 +const N02_URL: &str = "https://nlftp.mlit.go.jp/ksj/gml/data/N02/N02-25/N02-25_GML.zip"; +const CACHE_DIR: &str = "data/N02-25"; +/// ZIP の中の線路区間。Shapefile と GeoJSON が Shift-JIS 版と UTF-8 版の両方で +/// 入っているが、読むのは UTF-8 の GeoJSON だけ。 +const GEOJSON_NAME: &str = "N02-25_RailroadSection.geojson"; + +/// `companies.company_name_h` から「株式会社」を除いた名前が N02 の事業者名 +/// (`N02_004`) と一致しない会社。公営は N02 では自治体名になる。 +/// +/// ここに無く名前も一致しない会社 (神戸高速鉄道など、自社で列車を走らせない +/// 会社) は、事業者で絞らずに全事業者の線路で測る。 +const OPERATOR_ALIASES: &[(i32, &str)] = &[ + (16, "東急電鉄"), + (101, "札幌市"), + (102, "函館市"), + (107, "アイジーアールいわて銀河鉄道"), + (115, "仙台市"), + (119, "東京都"), + (130, "横浜市"), + (157, "上田電鉄"), + (170, "岳南電車"), + (176, "JR東海交通事業"), + (179, "名古屋市"), + (194, "WILLER\u{3000}TRAINS"), + (195, "京都市"), + (211, "神戸市"), + (228, "とさでん交通"), + (231, "福岡市"), + (241, "熊本市"), + (242, "鹿児島市"), + (252, "一般社団法人札幌市交通事業振興公社"), +]; + +/// 線路上の距離がこれを超える組は測れなかったものとして扱う。直線距離の 3 倍か、 +/// 直線距離 + 5km の大きい方。実在の最大は木次線の出雲坂根〜三井野原 +/// (三段スイッチバック) の約 3.6 倍。これより長い経路は、線形が途切れて +/// 別の路線を大回りしたものとみなす。 +fn max_track_length(straight: f64) -> f64 { + (3.0 * straight).max(straight + 5_000.0) +} + +/// 環状運転の継ぎ目 (末尾駅 -> 先頭駅) も組にする直線距離の上限。 +/// `arrival_estimation::is_circular_route` の上限と揃える。 +const SEAM_MAX_METERS: f64 = 3_000.0; +const SEAM_MIN_STATIONS: usize = 6; + +pub fn load() -> Result { + let path = Path::new(CACHE_DIR).join(GEOJSON_NAME); + if path.is_file() { + info!("国土数値情報 (鉄道) は取得済みなのでダウンロードを省略する"); + } else { + download(&path)?; + } + + #[derive(Deserialize)] + struct FeatureCollection { + features: Vec, + } + #[derive(Deserialize)] + struct Feature { + properties: Properties, + geometry: Geometry, + } + #[derive(Deserialize)] + struct Properties { + #[serde(rename = "N02_004")] + operator: String, + } + /// 線路区間はすべて LineString。 + #[derive(Deserialize)] + struct Geometry { + coordinates: Vec<[f64; 2]>, + } + + let file = File::open(&path).with_context(|| format!("{} を開けない", path.display()))?; + let collection: FeatureCollection = serde_json::from_reader(BufReader::new(file)) + .with_context(|| format!("{} を読めない", path.display()))?; + let sections: Vec
= collection + .features + .into_iter() + .map(|f| Section { + operator: f.properties.operator, + coordinates: f.geometry.coordinates, + }) + .collect(); + let network = RailNetwork::build(§ions); + info!( + "国土数値情報 (鉄道): 線路区間 {} / 頂点 {} / 辺 {}", + sections.len(), + network.vertex_count(), + network.edge_count() + ); + Ok(network) +} + +/// ZIP を取得して線路区間の GeoJSON だけを取り出す。一時ファイルへ書いてから +/// 置き換えるので、途中で落ちても壊れたキャッシュは残らない。 +fn download(path: &Path) -> Result<()> { + info!("国土数値情報 (鉄道) を取得する"); + // blocking クライアントの既定は本文の受信まで含めて 30 秒。約 13MB の ZIP を + // 国の配信サーバーから落とすには短く、取得の失敗は preprocessor の失敗になる。 + let client = reqwest::blocking::Client::builder() + .connect_timeout(Duration::from_secs(30)) + .timeout(Duration::from_secs(600)) + .build()?; + let response = client.get(N02_URL).send()?; + if !response.status().is_success() { + bail!( + "国土数値情報 (鉄道) の取得に失敗: HTTP {}", + response.status() + ); + } + let bytes = response.bytes()?; + + let mut archive = ZipArchive::new(Cursor::new(bytes))?; + let name = archive + .file_names() + .find(|name| name.ends_with(&format!("UTF-8/{GEOJSON_NAME}"))) + .map(str::to_string) + .with_context(|| format!("ZIP に UTF-8/{GEOJSON_NAME} が無い"))?; + let mut contents = Vec::new(); + archive.by_name(&name)?.read_to_end(&mut contents)?; + + fs::create_dir_all(CACHE_DIR)?; + let temporary = path.with_extension("geojson.tmp"); + fs::write(&temporary, &contents)?; + fs::rename(&temporary, path)?; + info!("{} を書き出した", path.display()); + Ok(()) +} + +struct StationPoint { + station_g_cd: i32, + line_cd: i32, + active: bool, + lat: f64, + lon: f64, +} + +/// 隣り合う駅の組ごとの線路の長さを `dataset.connections` へ書き出す。 +/// +/// `dataset.connections` には `data/8!connections.csv` の行 (手で直した値) が +/// 読み込まれており、同じ組ではそちらを残す。 +pub fn generate_connections(dataset: &mut Dataset, network: &RailNetwork) -> Result<()> { + let stations = rail_stations(dataset); + let pairs = adjacent_pairs(dataset, &stations); + let operators = line_operators(dataset, network); + + let c_station1 = dataset.connections.col("station_cd1"); + let c_station2 = dataset.connections.col("station_cd2"); + let c_distance = dataset.connections.col("distance"); + + // 手で書いた値。距離の書式は揃えて整数メートルにする。 + let mut distances: HashMap<(i32, i32), i64> = HashMap::new(); + for row in dataset.connections.rows() { + let (Some(a), Some(b)) = (cell_i32(row, c_station1), cell_i32(row, c_station2)) else { + bail!("8!connections.csv に駅コードの無い行がある"); + }; + let Some(distance) = row[c_distance] + .as_deref() + .and_then(|v| v.trim().parse::().ok()) + .filter(|v| v.is_finite() && *v >= 0.0) + else { + bail!("8!connections.csv の {a} - {b} の distance が 0 以上の数値ではない"); + }; + distances.insert(ordered(a, b), distance.round() as i64); + } + let manual = distances.len(); + + let (mut computed, mut same_group, mut fallback) = (0, 0, 0); + // 測れなかった組。廃線の駅は N02 に線路が無いので、営業中の駅どうしを分けて数える。 + let (mut missing, mut missing_active) = (0, 0); + for &(a, b) in &pairs { + if distances.contains_key(&(a, b)) { + continue; + } + let (sa, sb) = (&stations[&a], &stations[&b]); + // 直通運転の境界駅は、同じ駅グループの 2 つの駅が系統の中で隣り合う。 + if sa.station_g_cd == sb.station_g_cd { + distances.insert((a, b), 0); + same_group += 1; + continue; + } + + let straight = haversine_distance(sa.lat, sa.lon, sb.lat, sb.lon); + let limit = max_track_length(straight); + // まず両駅の路線の事業者の線路だけで測る。他社線を走る路線 + // (北陸新幹線の上越妙高以西、相鉄・JR直通線など) はそれで線路に寄せられない + // ので、全事業者の線路で測り直す。 + let own: Option> = + match (operators.get(&sa.line_cd), operators.get(&sb.line_cd)) { + (Some(&x), Some(&y)) => Some([x, y].into_iter().collect()), + _ => None, + }; + let mut length = own.as_ref().and_then(|ops| { + network.track_length((sa.lat, sa.lon), (sb.lat, sb.lon), Some(ops), limit) + }); + if length.is_none() { + length = network.track_length((sa.lat, sa.lon), (sb.lat, sb.lon), None, limit); + if length.is_some() && own.is_some() { + fallback += 1; + } + } + match length { + // 線路の長さは直線距離より短くならない。短く出るのは、駅の座標と線路の + // 位置がずれて寄せた位置どうしが近づいたときなので、直線距離で抑える。 + Some(length) => { + distances.insert((a, b), length.max(straight).round() as i64); + computed += 1; + } + None => { + missing += 1; + if sa.active && sb.active { + missing_active += 1; + } + } + } + } + + let mut rows: Vec<((i32, i32), i64)> = distances.into_iter().collect(); + rows.sort_unstable(); + let blank = dataset.connections.blank_row(); + let c_id = dataset.connections.col("id"); + dataset.connections.clear(); + for (i, ((a, b), distance)) in rows.into_iter().enumerate() { + let mut row = blank.clone(); + row[c_id] = int(i as i32 + 1); + row[c_station1] = int(a); + row[c_station2] = int(b); + row[c_distance] = text(distance.to_string()); + dataset.connections.push(row); + } + + info!( + "線路の長さ: 駅の組 {} / 計算 {computed} (うち全事業者で測り直し {fallback}) / \ + 同じ駅グループ {same_group} / 手入力 {manual} / 測れず {missing} (うち営業中の駅どうし {missing_active})", + pairs.len() + ); + if missing_active > pairs.len() / 100 { + warn!( + "営業中の駅どうしで線路の長さを測れなかった組が {missing_active} 件ある。\ + N02 の版や OPERATOR_ALIASES を確認すること" + ); + } + Ok(()) +} + +fn ordered(a: i32, b: i32) -> (i32, i32) { + (a.min(b), a.max(b)) +} + +/// 鉄道駅 (バス停を除く) を station_cd で引けるようにする。 +fn rail_stations(dataset: &Dataset) -> HashMap { + let t = &dataset.stations; + let (c_cd, c_g_cd, c_line, c_status, c_transport) = ( + t.col("station_cd"), + t.col("station_g_cd"), + t.col("line_cd"), + t.col("e_status"), + t.col("transport_type"), + ); + let (c_lat, c_lon) = (t.col("lat"), t.col("lon")); + let coordinate = |row: &[Option], idx: usize| -> Option { + row[idx].as_deref().and_then(|v| v.trim().parse().ok()) + }; + t.rows() + .iter() + .filter(|row| cell_i32(row, c_transport) == Some(0)) + .filter_map(|row| { + Some(( + cell_i32(row, c_cd)?, + StationPoint { + station_g_cd: cell_i32(row, c_g_cd)?, + line_cd: cell_i32(row, c_line)?, + active: cell_i32(row, c_status) == Some(0), + lat: coordinate(row, c_lat)?, + lon: coordinate(row, c_lon)?, + }, + )) + }) + .collect() +} + +/// API が返す駅の並びで隣り合う組 (小さい station_cd が先)。 +/// +/// 並びは路線の (e_sort, station_cd) 順 (`lineStations` で系統を選べない路線) と、 +/// 系統の sst.id 順 (`lineGroupStations` / `trainRoute` / 系統を選べた +/// `lineStations`)。Worker は廃止駅 (e_status != 0) を除いて返すので、除いた並びと +/// 除かない並びの両方から組を取る。環状運転の系統は末尾駅から先頭駅へ戻る組も取る +/// (`trainRoute` は継ぎ目をまたいで切り出す)。 +fn adjacent_pairs(dataset: &Dataset, stations: &HashMap) -> Vec<(i32, i32)> { + let mut sequences: Vec> = Vec::new(); + + let t = &dataset.stations; + let (c_cd, c_line, c_sort) = (t.col("station_cd"), t.col("line_cd"), t.col("e_sort")); + let mut by_line: HashMap> = HashMap::new(); + for row in t.rows() { + let (Some(cd), Some(line)) = (cell_i32(row, c_cd), cell_i32(row, c_line)) else { + continue; + }; + if stations.contains_key(&cd) { + by_line + .entry(line) + .or_default() + .push((cell_i32(row, c_sort).unwrap_or(0), cd)); + } + } + for mut line in by_line.into_values() { + line.sort_unstable(); + sequences.push(line.into_iter().map(|(_, cd)| cd).collect()); + } + + // sst は id 順 (= 停車順) に並んでいる。 + let s = &dataset.sst; + let (c_station, c_group) = (s.col("station_cd"), s.col("line_group_cd")); + let mut by_group: HashMap> = HashMap::new(); + for row in s.rows() { + let (Some(cd), Some(group)) = (cell_i32(row, c_station), cell_i32(row, c_group)) else { + continue; + }; + if stations.contains_key(&cd) { + by_group.entry(group).or_default().push(cd); + } + } + sequences.extend(by_group.into_values()); + + let mut pairs: HashSet<(i32, i32)> = HashSet::new(); + for all in sequences { + let active: Vec = all + .iter() + .copied() + .filter(|cd| stations[cd].active) + .collect(); + for sequence in [all, active] { + for w in sequence.windows(2) { + if w[0] != w[1] { + pairs.insert(ordered(w[0], w[1])); + } + } + if let (true, Some(&first), Some(&last)) = ( + sequence.len() >= SEAM_MIN_STATIONS, + sequence.first(), + sequence.last(), + ) { + let (f, l) = (&stations[&first], &stations[&last]); + if first != last + && haversine_distance(f.lat, f.lon, l.lat, l.lon) <= SEAM_MAX_METERS + { + pairs.insert(ordered(first, last)); + } + } + } + } + + let mut pairs: Vec<(i32, i32)> = pairs.into_iter().collect(); + pairs.sort_unstable(); + pairs +} + +/// 路線 -> その路線の会社の、N02 での事業者番号。N02 に無い会社の路線は入らない。 +fn line_operators(dataset: &Dataset, network: &RailNetwork) -> HashMap { + let c = &dataset.companies; + let (c_cd, c_name) = (c.col("company_cd"), c.col("company_name_h")); + let aliases: HashMap = OPERATOR_ALIASES.iter().copied().collect(); + let company_operator: HashMap = c + .rows() + .iter() + .filter_map(|row| { + let cd = cell_i32(row, c_cd)?; + let name = match aliases.get(&cd) { + Some(alias) => (*alias).to_string(), + None => row[c_name] + .as_deref()? + .replace("株式会社", "") + .trim() + .to_string(), + }; + Some((cd, network.operator_id(&name)?)) + }) + .collect(); + + let l = &dataset.lines; + let (l_cd, l_company) = (l.col("line_cd"), l.col("company_cd")); + l.rows() + .iter() + .filter_map(|row| { + let operator = company_operator.get(&cell_i32(row, l_company)?)?; + Some((cell_i32(row, l_cd)?, *operator)) + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::rail::{ + ALIAS_COLUMNS, COMPANY_COLUMNS, CONNECTION_COLUMNS, LINE_ALIAS_COLUMNS, LINE_COLUMNS, + SST_COLUMNS, STATION_COLUMNS, TYPE_COLUMNS, + }; + use crate::table::Table; + + fn push(table: &mut Table, cells: &[(&str, &str)]) { + let mut row = table.blank_row(); + for (column, value) in cells { + row[table.col(column)] = text(*value); + } + table.push(row); + } + + /// 経度 139.00〜139.03 を東西に走る「テスト鉄道」の線路と、その上の駅。 + /// + /// - 路線 10: 101 (139.00) - 102 (139.01) - 103 (139.02) - 104 (線路から約 2.2km 北) + /// - 路線 20: 201 (103 と同じ駅グループ) + /// - 系統 900: 101 - 102 - 103 - 201 (直通の境界駅で同じ駅グループが隣り合う) + fn fixture() -> (Dataset, RailNetwork) { + let mut dataset = Dataset { + companies: Table::new(COMPANY_COLUMNS, Some("company_cd")), + lines: Table::new(LINE_COLUMNS, Some("line_cd")), + stations: Table::new(STATION_COLUMNS, Some("station_cd")), + types: Table::new(TYPE_COLUMNS, Some("type_cd")), + sst: Table::new(SST_COLUMNS, None), + aliases: Table::new(ALIAS_COLUMNS, Some("id")), + line_aliases: Table::new(LINE_ALIAS_COLUMNS, Some("id")), + connections: Table::new(CONNECTION_COLUMNS, None), + }; + push( + &mut dataset.companies, + &[ + ("company_cd", "1"), + ("company_name_h", "テスト鉄道株式会社"), + ], + ); + for line in ["10", "20"] { + push( + &mut dataset.lines, + &[("line_cd", line), ("company_cd", "1")], + ); + } + for (cd, g_cd, line, sort, lat, lon) in [ + ("101", "101", "10", "1", "35.0", "139.00"), + ("102", "102", "10", "2", "35.0", "139.01"), + ("103", "103", "10", "3", "35.0", "139.02"), + ("104", "104", "10", "4", "35.02", "139.03"), + ("201", "103", "20", "1", "35.0", "139.02"), + ] { + push( + &mut dataset.stations, + &[ + ("station_cd", cd), + ("station_g_cd", g_cd), + ("line_cd", line), + ("e_sort", sort), + ("e_status", "0"), + ("transport_type", "0"), + ("lat", lat), + ("lon", lon), + ], + ); + } + for cd in ["101", "102", "103", "201"] { + push( + &mut dataset.sst, + &[ + ("station_cd", cd), + ("type_cd", "100"), + ("line_group_cd", "900"), + ], + ); + } + let network = RailNetwork::build(&[Section { + operator: "テスト鉄道".to_string(), + coordinates: vec![ + [139.00, 35.0], + [139.01, 35.0], + [139.02, 35.0], + [139.03, 35.0], + ], + }]); + (dataset, network) + } + + fn connections(dataset: &Dataset) -> Vec<(i32, i32, String)> { + let t = &dataset.connections; + let (a, b, d) = ( + t.col("station_cd1"), + t.col("station_cd2"), + t.col("distance"), + ); + t.rows() + .iter() + .map(|row| { + ( + cell_i32(row, a).unwrap(), + cell_i32(row, b).unwrap(), + row[d].clone().unwrap(), + ) + }) + .collect() + } + + #[test] + fn measures_adjacent_stations_and_keeps_manual_values() { + let (mut dataset, network) = fixture(); + // 手で直した値は向きを問わず計算結果より優先し、整数メートルに揃える + push( + &mut dataset.connections, + &[ + ("station_cd1", "102"), + ("station_cd2", "101"), + ("distance", "1234.4"), + ], + ); + + generate_connections(&mut dataset, &network).unwrap(); + + let straight = haversine_distance(35.0, 139.01, 35.0, 139.02).round(); + assert_eq!( + connections(&dataset), + vec![ + (101, 102, "1234".to_string()), + (102, 103, straight.to_string()), + // 104 は線路から遠いので測れない。同じ駅グループの 103 - 201 は 0 + (103, 201, "0".to_string()), + ] + ); + let ids: Vec> = dataset + .connections + .rows() + .iter() + .map(|row| cell_i32(row, dataset.connections.col("id"))) + .collect(); + assert_eq!(ids, vec![Some(1), Some(2), Some(3)]); + } + + #[test] + fn rejects_manual_rows_without_a_valid_distance() { + let (mut dataset, network) = fixture(); + push( + &mut dataset.connections, + &[ + ("station_cd1", "101"), + ("station_cd2", "102"), + ("distance", "-1"), + ], + ); + assert!(generate_connections(&mut dataset, &network).is_err()); + } + + #[test] + fn pairs_come_from_line_order_and_group_order() { + let (dataset, _) = fixture(); + let stations = rail_stations(&dataset); + assert_eq!( + adjacent_pairs(&dataset, &stations), + vec![(101, 102), (102, 103), (103, 104), (103, 201)] + ); + } +} diff --git a/preprocessor/src/track/network.rs b/preprocessor/src/track/network.rs new file mode 100644 index 00000000..a9da10a3 --- /dev/null +++ b/preprocessor/src/track/network.rs @@ -0,0 +1,468 @@ +//! 国土数値情報の鉄道データ (N02) の線路区間を、駅間の線路の長さを測るための +//! 無向グラフにする。 +//! +//! 線路区間は LineString の集まりで、端点を共有する区間どうしがつながっている。 +//! 頂点を座標で同一視してグラフにし、2 駅をそれぞれ近くの線路へ寄せて、その間の +//! 最短経路の長さを線路の長さとする。 + +use std::cmp::Ordering; +use std::collections::{BinaryHeap, HashMap, HashSet}; + +use stationapi::domain::arrival_estimation::haversine_distance; + +/// 駅を線路へ寄せるときに見る範囲 (メートル)。 +/// +/// 駅の座標は駅舎に置かれていることが多く、線路から数百メートル離れることがある +/// (湘南新宿ラインの武蔵小杉は南武線側の座標で、品鶴線から約 350m)。 +pub const SNAP_RADIUS_METERS: f64 = 500.0; + +/// 駅から線路までの距離に掛ける重み。 +/// +/// 範囲内の線路をすべて候補にして、「寄せた距離 × この重み + 線路上の距離」が +/// 最小になる組を選ぶ。最も近い線路だけに寄せると、大きな駅で同じ事業者の別路線 +/// (本町の御堂筋線と四つ橋線など) に寄ってしまい、乗換駅をまわる遠回りが出る。 +/// 重みを 1 以下にすると、相手の駅の近くまで寄せた方が安くなり、線路の長さが +/// 実際より短く出る。 +const SNAP_PENALTY: f64 = 2.0; + +/// 空間索引の升目 (度)。緯度 45 度でも経度方向に約 780m あり +/// [`SNAP_RADIUS_METERS`] より大きいので、周囲 9 升を見れば取りこぼさない。 +const CELL_DEGREES: f64 = 0.01; + +/// 頂点を同一視する座標の桁。N02 の座標は小数第 5 位 (約 1m) までのものが多い。 +const VERTEX_SCALE: f64 = 1e5; + +/// N02 の線路区間 1 本。 +pub struct Section { + /// 事業者名 (`N02_004`)。 + pub operator: String, + /// `[経度, 緯度]` の並び。 + pub coordinates: Vec<[f64; 2]>, +} + +struct Edge { + a: u32, + b: u32, + operator: u16, + length: f64, +} + +/// 駅を線路へ寄せた位置。 +struct Snap { + edge: u32, + /// 辺の上の位置。0 が `a`、1 が `b`。 + t: f64, + /// 駅から寄せた位置までの距離 (メートル)。 + offset: f64, +} + +pub struct RailNetwork { + lat: Vec, + lon: Vec, + edges: Vec, + /// 頂点 -> 接する辺 (CSR)。頂点 `v` の辺は `adj[adj_start[v]..adj_start[v + 1]]`。 + adj_start: Vec, + adj: Vec, + /// 升目 -> その升目にかかる辺。 + grid: HashMap<(i32, i32), Vec>, + operators: HashMap, +} + +impl RailNetwork { + pub fn build(sections: &[Section]) -> Self { + let mut vertex_of: HashMap<(i64, i64), u32> = HashMap::new(); + let mut lat: Vec = Vec::new(); + let mut lon: Vec = Vec::new(); + let mut edges: Vec = Vec::new(); + let mut operators: HashMap = HashMap::new(); + + for section in sections { + let next_operator = operators.len() as u16; + let operator = *operators + .entry(section.operator.clone()) + .or_insert(next_operator); + let mut previous: Option = None; + for &[x, y] in §ion.coordinates { + let key = ( + (x * VERTEX_SCALE).round() as i64, + (y * VERTEX_SCALE).round() as i64, + ); + let vertex = *vertex_of.entry(key).or_insert_with(|| { + lat.push(y); + lon.push(x); + (lat.len() - 1) as u32 + }); + if let Some(prev) = previous.filter(|&prev| prev != vertex) { + let length = haversine_distance( + lat[prev as usize], + lon[prev as usize], + lat[vertex as usize], + lon[vertex as usize], + ); + edges.push(Edge { + a: prev, + b: vertex, + operator, + length, + }); + } + previous = Some(vertex); + } + } + + let mut adj_start = vec![0usize; lat.len() + 1]; + for edge in &edges { + adj_start[edge.a as usize + 1] += 1; + adj_start[edge.b as usize + 1] += 1; + } + for i in 1..adj_start.len() { + adj_start[i] += adj_start[i - 1]; + } + let mut fill = adj_start.clone(); + let mut adj = vec![0u32; adj_start[lat.len()]]; + for (i, edge) in edges.iter().enumerate() { + for v in [edge.a, edge.b] { + adj[fill[v as usize]] = i as u32; + fill[v as usize] += 1; + } + } + + let mut grid: HashMap<(i32, i32), Vec> = HashMap::new(); + for (i, edge) in edges.iter().enumerate() { + let (a, b) = (edge.a as usize, edge.b as usize); + let (y0, y1) = (cell(lat[a].min(lat[b])), cell(lat[a].max(lat[b]))); + let (x0, x1) = (cell(lon[a].min(lon[b])), cell(lon[a].max(lon[b]))); + for y in y0..=y1 { + for x in x0..=x1 { + grid.entry((y, x)).or_default().push(i as u32); + } + } + } + + RailNetwork { + lat, + lon, + edges, + adj_start, + adj, + grid, + operators, + } + } + + pub fn vertex_count(&self) -> usize { + self.lat.len() + } + + pub fn edge_count(&self) -> usize { + self.edges.len() + } + + /// 事業者名から事業者の番号を引く。N02 に無い事業者は `None`。 + pub fn operator_id(&self, name: &str) -> Option { + self.operators.get(name).copied() + } + + /// `from` と `to` のあいだの線路の長さ (メートル)。 + /// + /// `operators` を渡すと、その事業者の線路だけを使う。線路上の距離が + /// `max_length` を超える場合と、どちらかの駅の近くに線路が無い場合は `None`。 + pub fn track_length( + &self, + from: (f64, f64), + to: (f64, f64), + operators: Option<&HashSet>, + max_length: f64, + ) -> Option { + let origin = self.snaps(from, operators); + let destination = self.snaps(to, operators); + if origin.is_empty() || destination.is_empty() { + return None; + } + + // (費用, 線路上の距離)。費用は寄せた距離の重みを含み、経路の選択にだけ使う。 + let mut best: Option<(f64, f64)> = None; + + // 同じ辺に寄せた場合は、辺の上の位置の差がそのまま長さになる。 + for o in &origin { + for d in destination.iter().filter(|d| d.edge == o.edge) { + let length = (o.t - d.t).abs() * self.edges[o.edge as usize].length; + improve( + &mut best, + SNAP_PENALTY * (o.offset + d.offset) + length, + length, + ); + } + } + + let starts = self.snap_ends(&origin); + let goals = self.snap_ends(&destination); + + let mut dist: HashMap = HashMap::new(); + let mut heap = BinaryHeap::new(); + for (&vertex, &(cost, length)) in &starts { + dist.insert(vertex, cost); + heap.push(Label { + cost, + length, + vertex, + }); + } + + let cost_limit = max_length + 2.0 * SNAP_PENALTY * SNAP_RADIUS_METERS; + while let Some(Label { + cost, + length, + vertex, + }) = heap.pop() + { + if dist.get(&vertex).is_some_and(|&d| cost > d) { + continue; + } + if cost > cost_limit || best.is_some_and(|(best_cost, _)| cost >= best_cost) { + break; + } + if let Some(&(goal_cost, goal_length)) = goals.get(&vertex) { + improve(&mut best, cost + goal_cost, length + goal_length); + } + let v = vertex as usize; + for &e in &self.adj[self.adj_start[v]..self.adj_start[v + 1]] { + let edge = &self.edges[e as usize]; + if operators.is_some_and(|ops| !ops.contains(&edge.operator)) { + continue; + } + let next = if edge.a == vertex { edge.b } else { edge.a }; + let next_cost = cost + edge.length; + if dist.get(&next).is_none_or(|&d| next_cost < d) { + dist.insert(next, next_cost); + heap.push(Label { + cost: next_cost, + length: length + edge.length, + vertex: next, + }); + } + } + } + + best.map(|(_, length)| length) + .filter(|&length| length <= max_length) + } + + /// 駅から [`SNAP_RADIUS_METERS`] 以内にある辺と、その辺の上で最も駅に近い位置。 + fn snaps(&self, (lat, lon): (f64, f64), operators: Option<&HashSet>) -> Vec { + let (y0, x0) = (cell(lat), cell(lon)); + let mut seen: HashSet = HashSet::new(); + let mut out = Vec::new(); + for y in y0 - 1..=y0 + 1 { + for x in x0 - 1..=x0 + 1 { + let Some(edges) = self.grid.get(&(y, x)) else { + continue; + }; + for &e in edges { + if !seen.insert(e) { + continue; + } + if operators.is_some_and(|ops| !ops.contains(&self.edges[e as usize].operator)) + { + continue; + } + let (offset, t) = self.project(lat, lon, e); + if offset <= SNAP_RADIUS_METERS { + out.push(Snap { edge: e, t, offset }); + } + } + } + } + out + } + + /// 寄せた位置から辺の両端までを、端点ごとに最も安いものだけ残す。 + /// 値は (費用, 線路上の距離)。 + fn snap_ends(&self, snaps: &[Snap]) -> HashMap { + let mut ends: HashMap = HashMap::new(); + for snap in snaps { + let edge = &self.edges[snap.edge as usize]; + for (vertex, length) in [ + (edge.a, snap.t * edge.length), + (edge.b, (1.0 - snap.t) * edge.length), + ] { + let cost = SNAP_PENALTY * snap.offset + length; + let entry = ends.entry(vertex).or_insert((f64::INFINITY, 0.0)); + if cost < entry.0 { + *entry = (cost, length); + } + } + } + ends + } + + /// 駅を辺へ垂直に下ろした位置。駅間程度の範囲なので、経度を緯度の余弦で縮めた + /// 平面で近似する。返り値は (駅からの距離 (メートル), 辺の上の位置)。 + fn project(&self, lat: f64, lon: f64, e: u32) -> (f64, f64) { + let edge = &self.edges[e as usize]; + let (a, b) = (edge.a as usize, edge.b as usize); + let k = lat.to_radians().cos(); + let (ax, ay) = (self.lon[a] * k, self.lat[a]); + let (dx, dy) = (self.lon[b] * k - ax, self.lat[b] - ay); + let len2 = dx * dx + dy * dy; + let t = if len2 == 0.0 { + 0.0 + } else { + (((lon * k - ax) * dx + (lat - ay) * dy) / len2).clamp(0.0, 1.0) + }; + let (qlat, qlon) = (ay + t * dy, (ax + t * dx) / k); + (haversine_distance(lat, lon, qlat, qlon), t) + } +} + +/// (費用, 線路上の距離) の候補を、費用が小さければ採る。 +fn improve(best: &mut Option<(f64, f64)>, cost: f64, length: f64) { + if best.is_none_or(|(best_cost, _)| cost < best_cost) { + *best = Some((cost, length)); + } +} + +fn cell(degrees: f64) -> i32 { + (degrees / CELL_DEGREES).floor() as i32 +} + +/// Dijkstra の探索ラベル。`BinaryHeap` は最大ヒープなので費用の比較を逆にする。 +/// 費用が同じときも順序が決まるよう、頂点と長さでも比べる。 +struct Label { + cost: f64, + length: f64, + vertex: u32, +} + +impl PartialEq for Label { + fn eq(&self, other: &Self) -> bool { + self.cmp(other) == Ordering::Equal + } +} + +impl Eq for Label {} + +impl PartialOrd for Label { + fn partial_cmp(&self, other: &Self) -> Option { + Some(self.cmp(other)) + } +} + +impl Ord for Label { + fn cmp(&self, other: &Self) -> Ordering { + other + .cost + .total_cmp(&self.cost) + .then_with(|| other.vertex.cmp(&self.vertex)) + .then_with(|| other.length.total_cmp(&self.length)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn section(operator: &str, coordinates: &[[f64; 2]]) -> Section { + Section { + operator: operator.to_string(), + coordinates: coordinates.to_vec(), + } + } + + /// 経度 139.00 から東へ延びる線路。緯度 35 度で経度 0.01 度は約 911m。 + fn straight() -> Vec
{ + vec![section( + "A", + &[ + [139.00, 35.0], + [139.01, 35.0], + [139.02, 35.0], + [139.03, 35.0], + ], + )] + } + + #[test] + fn measures_along_the_track() { + let network = RailNetwork::build(&straight()); + let length = network + .track_length((35.0, 139.005), (35.0, 139.025), None, 10_000.0) + .unwrap(); + let expected = haversine_distance(35.0, 139.005, 35.0, 139.025); + assert!((length - expected).abs() < 1.0, "{length} vs {expected}"); + } + + #[test] + fn follows_the_curve_instead_of_the_straight_line() { + // 北へ 0.01 度 (約 1.1km) 迂回してから戻る線路。 + let sections = vec![section( + "A", + &[ + [139.00, 35.0], + [139.00, 35.01], + [139.01, 35.01], + [139.01, 35.0], + ], + )]; + let network = RailNetwork::build(§ions); + let length = network + .track_length((35.0, 139.00), (35.0, 139.01), None, 10_000.0) + .unwrap(); + let straight = haversine_distance(35.0, 139.00, 35.0, 139.01); + assert!(length > 2.0 * straight, "{length} vs {straight}"); + } + + #[test] + fn stations_far_from_any_track_have_no_length() { + let network = RailNetwork::build(&straight()); + // 線路から北へ約 1.1km + assert!(network + .track_length((35.01, 139.005), (35.0, 139.025), None, 10_000.0) + .is_none()); + } + + #[test] + fn prefers_the_track_both_stations_share() { + // 駅 X は線路 B のすぐ横にあるが、線路 A からも 200m ほどの位置にある。 + // B は Y の近くを通らず、A とは 3km 先でつながる。A に寄せて測るべき。 + let sections = vec![ + section("A", &[[139.000, 35.0], [139.010, 35.0]]), + section( + "B", + &[ + [139.000, 35.0018], + [139.030, 35.0018], + [139.030, 35.0], + [139.010, 35.0], + ], + ), + ]; + let network = RailNetwork::build(§ions); + let length = network + .track_length((35.0018, 139.002), (35.0, 139.008), None, 10_000.0) + .unwrap(); + let along_a = haversine_distance(35.0, 139.002, 35.0, 139.008); + assert!((length - along_a).abs() < 1.0, "{length} vs {along_a}"); + } + + #[test] + fn operator_filter_excludes_other_tracks() { + let network = RailNetwork::build(&straight()); + let other: HashSet = [99].into_iter().collect(); + assert!(network + .track_length((35.0, 139.005), (35.0, 139.025), Some(&other), 10_000.0) + .is_none()); + let own: HashSet = [network.operator_id("A").unwrap()].into_iter().collect(); + assert!(network + .track_length((35.0, 139.005), (35.0, 139.025), Some(&own), 10_000.0) + .is_some()); + } + + #[test] + fn gives_up_beyond_the_length_limit() { + let network = RailNetwork::build(&straight()); + assert!(network + .track_length((35.0, 139.0), (35.0, 139.03), None, 1_000.0) + .is_none()); + } +} diff --git a/schema/public.graphql b/schema/public.graphql index 433a51b1..8907141e 100644 --- a/schema/public.graphql +++ b/schema/public.graphql @@ -137,6 +137,7 @@ type Station { stationNumbers: [StationNumber!] stopCondition: StopCondition distance: Float + trackDistanceFromPrevious: Float hasTrainTypes: Boolean trainType: TrainTypeNested lines: [LineNested!] @@ -167,6 +168,7 @@ type StationNested { stationNumbers: [StationNumber!] stopCondition: StopCondition distance: Float + trackDistanceFromPrevious: Float hasTrainTypes: Boolean trainType: TrainTypeNested lines: [LineNested!] diff --git a/src/graphql/types.rs b/src/graphql/types.rs index 10797739..fd4589df 100644 --- a/src/graphql/types.rs +++ b/src/graphql/types.rs @@ -135,6 +135,7 @@ macro_rules! define_station { pub station_numbers: Option>, pub stop_condition: Option, pub distance: Option, + pub track_distance_from_previous: Option, pub has_train_types: Option, pub train_type: Option>, pub lines: Option>, @@ -167,6 +168,7 @@ macro_rules! define_station { station_numbers: Some(v.station_numbers.into_iter().map(Into::into).collect()), stop_condition: Some(StopCondition::from(v.stop_condition)), distance: v.distance, + track_distance_from_previous: v.track_distance_from_previous, has_train_types: v.has_train_types, train_type: v.train_type.map(|t| Box::new((*t).into())), lines: Some(v.lines.into_iter().map(Into::into).collect()), diff --git a/src/index.rs b/src/index.rs index 487547c4..02f354b8 100644 --- a/src/index.rs +++ b/src/index.rs @@ -142,6 +142,7 @@ impl StationRecord { e_sort: self.e_sort, stop_condition: StopCondition::All, distance: None, + track_distance_from_previous: None, train_type: None, has_train_types: false, company_cd: line.map(|l| l.company_cd), @@ -1035,6 +1036,31 @@ pub fn sst_by_group(line_group_cd: i32) -> impl Iterator Option { + let rows = CONNECTIONS_BIN.as_chunks::<12>().0; + let read = |row: &[u8; 12], i: usize| -> i32 { + i32::from_le_bytes([row[i * 4], row[i * 4 + 1], row[i * 4 + 2], row[i * 4 + 3]]) + }; + let key = (a.min(b), a.max(b)); + rows.binary_search_by(|row| (read(row, 0), read(row, 1)).cmp(&key)) + .ok() + .map(|i| read(&rows[i], 2) as f64) +} + /// line_cd -> line_name_rn。 /// Line エンティティは line_name_rn を持たない (検索専用列) ため別に保持する。 static LINE_NAME_RN: OnceLock> = OnceLock::new(); @@ -1263,6 +1289,29 @@ mod tests { } } + /// 線路の長さのバイナリは (小さい駅, 大きい駅) の昇順で重複が無く、 + /// どちらの向きで引いても同じ値になる。並びが崩れると二分探索が取りこぼす。 + #[test] + fn track_distances_are_sorted_and_symmetric() { + let rows = CONNECTIONS_BIN.as_chunks::<12>(); + assert!( + rows.1.is_empty(), + "connections.bin が 12 バイトの倍数ではない" + ); + let read = |row: &[u8; 12], i: usize| { + i32::from_le_bytes(row[i * 4..i * 4 + 4].try_into().unwrap()) + }; + let keys: Vec<(i32, i32)> = rows.0.iter().map(|r| (read(r, 0), read(r, 1))).collect(); + assert!(keys.iter().all(|(a, b)| a < b)); + assert!(keys.windows(2).all(|w| w[0] < w[1])); + for row in rows.0 { + let (a, b, distance) = (read(row, 0), read(row, 1), read(row, 2) as f64); + assert_eq!(track_distance(a, b), Some(distance)); + assert_eq!(track_distance(b, a), Some(distance)); + } + assert_eq!(track_distance(-1, -2), None); + } + /// グリッド索引は全件走査と同じ結果を返す。 /// 索引の絞り込みが範囲を取りこぼすと最近傍が欠けるため、実データで突き合わせる。 #[test] diff --git a/src/repository.rs b/src/repository.rs index 7a84228f..1ebbbf97 100644 --- a/src/repository.rs +++ b/src/repository.rs @@ -199,6 +199,16 @@ impl StationRepository for MemStationRepository { Ok(Arc::clone(route_network())) } + async fn get_track_distances( + &self, + pairs: &[(u32, u32)], + ) -> Result>, DomainError> { + Ok(pairs + .iter() + .map(|&(a, b)| index::track_distance(a as i32, b as i32)) + .collect()) + } + async fn get_by_coordinates( &self, latitude: f64, diff --git a/stationapi/src/domain/arrival_estimation.rs b/stationapi/src/domain/arrival_estimation.rs index 4d8672c0..60284185 100644 --- a/stationapi/src/domain/arrival_estimation.rs +++ b/stationapi/src/domain/arrival_estimation.rs @@ -739,6 +739,7 @@ mod tests { e_sort: station_cd, stop_condition: StopCondition::All, distance: None, + track_distance_from_previous: None, has_train_types: false, train_type: None, company_cd: Some(1), diff --git a/stationapi/src/domain/entity/station.rs b/stationapi/src/domain/entity/station.rs index 154da2e9..0572ae5a 100644 --- a/stationapi/src/domain/entity/station.rs +++ b/stationapi/src/domain/entity/station.rs @@ -36,6 +36,9 @@ pub struct Station { pub e_sort: i32, pub stop_condition: StopCondition, pub distance: Option, + /// 返す駅の並びで直前にある駅からの線路の長さ (メートル)。`distance` と同じく + /// 問い合わせごとに決まる値で、並びを返す問い合わせだけが埋める。 + pub track_distance_from_previous: Option, pub has_train_types: bool, pub train_type: Option>, // 路線から引く値 @@ -176,6 +179,7 @@ impl Station { e_sort, stop_condition, distance, + track_distance_from_previous: None, has_train_types, train_type, company_cd, diff --git a/stationapi/src/domain/repository/station_repository.rs b/stationapi/src/domain/repository/station_repository.rs index 818c19fb..36cacbe3 100644 --- a/stationapi/src/domain/repository/station_repository.rs +++ b/stationapi/src/domain/repository/station_repository.rs @@ -74,6 +74,21 @@ pub trait StationRepository: Send + Sync + 'static { "route network is not supported by this repository".to_string(), )) } + /// 駅の組ごとの、2 駅のあいだの線路の長さ (メートル)。組の向きは問わず、 + /// 結果は `pairs` と同じ並び。長さを持たない組は `None`。 + /// + /// 長さを持つのは、API が返す駅の並び (路線・系統) で隣り合う組だけ。 + /// + /// 既定はすべて `None`。線路の長さは駅に添える補足の値で、`None` は + /// 「データが無いので直線距離で代える」という正常な応答として決めてある + /// (`get_route_network` の既定がエラーなのは、空の網が「経路なし」という + /// 誤った答えになるため。こちらはそうならない)。 + async fn get_track_distances( + &self, + pairs: &[(u32, u32)], + ) -> Result>, DomainError> { + Ok(vec![None; pairs.len()]) + } /// 各座標から `radius_meters` 以内のバス停を、近い順に最大 /// `limit_per_station` 件返す。半径の外は呼び出し側でも採用されないため、 /// ここで切っておく (全国の最寄り N 件を作ってから捨てると、駅数に比例して diff --git a/stationapi/src/domain/route_search.rs b/stationapi/src/domain/route_search.rs index 56cf5cb1..f7c80dd1 100644 --- a/stationapi/src/domain/route_search.rs +++ b/stationapi/src/domain/route_search.rs @@ -818,6 +818,7 @@ mod tests { e_sort: sst_id, stop_condition: StopCondition::All, distance: None, + track_distance_from_previous: None, has_train_types: true, train_type: None, company_cd: Some(1), diff --git a/stationapi/src/model.rs b/stationapi/src/model.rs index fb6a7bd6..94eb757b 100644 --- a/stationapi/src/model.rs +++ b/stationapi/src/model.rs @@ -292,6 +292,10 @@ pub struct Station { /// [`StopCondition`] pub stop_condition: i32, pub distance: Option, + /// 返す駅の並びで直前にある駅からの線路の長さ (メートル)。国土数値情報の + /// 線路から求めた値で、直線距離ではない。先頭の駅、線路のデータが無い区間、 + /// 並びを返さない問い合わせでは `None`。 + pub track_distance_from_previous: Option, pub has_train_types: Option, pub train_type: Option>, /// [`TransportType`] diff --git a/stationapi/src/use_case/dto/station.rs b/stationapi/src/use_case/dto/station.rs index 03ed3f0b..39a3d3f1 100644 --- a/stationapi/src/use_case/dto/station.rs +++ b/stationapi/src/use_case/dto/station.rs @@ -48,6 +48,7 @@ impl From for ModelStation { .collect(), stop_condition: station.stop_condition.into(), distance: station.distance, + track_distance_from_previous: station.track_distance_from_previous, has_train_types: Some(station.has_train_types), train_type: station.train_type.map(|tt| Box::new((*tt).into())), transport_type: station.transport_type.into(), diff --git a/stationapi/src/use_case/interactor/query.rs b/stationapi/src/use_case/interactor/query.rs index ff6c6ab7..a14a5e2b 100644 --- a/stationapi/src/use_case/interactor/query.rs +++ b/stationapi/src/use_case/interactor/query.rs @@ -201,7 +201,7 @@ where .and_then(|sta| sta.line_group_cd), }; - let stations = self + let mut stations = self .update_station_vec_with_attributes( stations, line_group_id.map(|id| id as u32), @@ -209,6 +209,7 @@ where false, ) .await?; + self.attach_track_distances(&mut stations).await?; Ok(stations) } @@ -320,7 +321,7 @@ where .get_by_line_group_id(line_group_id) .await?; - let stations = self + let mut stations = self .update_station_vec_with_attributes( stations, Some(line_group_id), @@ -328,6 +329,7 @@ where false, ) .await?; + self.attach_track_distances(&mut stations).await?; Ok(stations) } @@ -1235,13 +1237,32 @@ where Ok(group_of) } + /// 返す駅の並びで隣り合う 2 駅のあいだの線路の長さを、後ろの駅の + /// `track_distance_from_previous` に入れる。先頭の駅は `None` のまま。 + /// + /// 並びが確定してから呼ぶこと (長さは並びの隣どうしで決まる)。 + async fn attach_track_distances(&self, stations: &mut [Station]) -> Result<(), UseCaseError> { + if stations.len() < 2 { + return Ok(()); + } + let pairs: Vec<(u32, u32)> = stations + .windows(2) + .map(|w| (w[0].station_cd as u32, w[1].station_cd as u32)) + .collect(); + let distances = self.station_repository.get_track_distances(&pairs).await?; + for (station, distance) in stations[1..].iter_mut().zip(distances) { + station.track_distance_from_previous = distance; + } + Ok(()) + } + /// 切り出した系統の駅列に付帯情報を付け、走行区間にする。 async fn train_route_segments( &self, sliced: Vec, line_group_id: u32, ) -> Result, UseCaseError> { - let sliced = self + let mut sliced = self .update_station_vec_with_attributes( sliced, Some(line_group_id), @@ -1249,6 +1270,7 @@ where false, ) .await?; + self.attach_track_distances(&mut sliced).await?; let mut segments: Vec = Vec::with_capacity(sliced.len()); // 経路スライス内で路線ごとに通過駅があるか。通過駅が無い路線では優等種別でも @@ -1808,6 +1830,7 @@ where e_sort: row.e_sort, stop_condition: row.stop_condition, distance: row.distance, + track_distance_from_previous: row.track_distance_from_previous, train_type, has_train_types: row.has_train_types, company_cd: row.company_cd, @@ -2055,6 +2078,7 @@ mod tests { e_sort: 1, stop_condition: StopCondition::All, distance: None, + track_distance_from_previous: None, has_train_types: false, train_type: None, company_cd: Some(1), @@ -5683,6 +5707,21 @@ mod tests { async fn get_by_line_group_id(&self, _: u32) -> Result, DomainError> { Ok(self.line_group_stations.clone()) } + /// station_cd が連続する組にだけ、組ごとに違う長さを返す + /// ((1000, 1001) は 100m、(1001, 1002) は 101m…)。(1005, 1006) は線路の + /// データが無い組とする。 + async fn get_track_distances( + &self, + pairs: &[(u32, u32)], + ) -> Result>, DomainError> { + Ok(pairs + .iter() + .map(|&(a, b)| { + let low = a.min(b); + (a.abs_diff(b) == 1 && low != 1005).then(|| f64::from(low - 900)) + }) + .collect()) + } async fn get_by_station_group_id_vec( &self, ids: &[u32], @@ -6026,6 +6065,68 @@ mod tests { assert!(stations.iter().all(|s| s.train_type.is_some())); } + /// 線路の長さは返す並びで直前にある駅からのもの。先頭の駅と、データの無い + /// 組は None + #[tokio::test] + async fn line_group_stations_carry_track_distances_in_order() { + let (interactor, _) = build_interactor(build_line_group(7)); + + let stations = interactor + .get_stations_by_line_group_id(1000, TransportTypeFilter::RailAndBus) + .await + .unwrap(); + + let distances: Vec> = stations + .iter() + .map(|s| s.track_distance_from_previous) + .collect(); + assert_eq!( + distances, + vec![ + None, + Some(100.0), + Some(101.0), + Some(102.0), + Some(103.0), + Some(104.0), + None + ] + ); + } + + #[tokio::test] + async fn line_stations_carry_track_distances() { + let interactor = build_line_interactor(build_typed_line_stations(3)); + + let stations = interactor + .get_stations_by_line_id(10, None, None, TransportTypeFilter::RailAndBus) + .await + .unwrap(); + + let distances: Vec> = stations + .iter() + .map(|s| s.track_distance_from_previous) + .collect(); + assert_eq!(distances, vec![None, Some(100.0), Some(101.0)]); + } + + /// 逆向きに切り出した区間では、進む向きで直前の駅からの長さになる + #[tokio::test] + async fn train_route_track_distances_follow_the_travel_direction() { + let (interactor, _) = build_interactor(build_line_group(20)); + + let segments = interactor + .get_train_route(1004, 1002, Some(1000)) + .await + .unwrap(); + + let distances: Vec> = segments + .iter() + .map(|s| s.station.as_ref().unwrap().track_distance_from_previous) + .collect(); + assert_eq!(distances, vec![None, Some(103.0), Some(102.0)]); + } + #[tokio::test] async fn enriches_only_the_requested_range() { let (interactor, calls) = build_interactor(build_line_group(20)); From c0e7fd001c62abbd39959bb4648f0e16404b3f6d Mon Sep 17 00:00:00 2001 From: Tsubasa SEKIGUCHI Date: Tue, 29 Sep 2026 08:44:12 +0900 Subject: [PATCH 2/2] =?UTF-8?q?stations(ids)=20=E3=81=A7=E3=82=82=E6=8C=87?= =?UTF-8?q?=E5=AE=9A=E3=81=97=E3=81=9F=20ID=20=E3=81=AE=E9=A0=86=E3=81=A7?= =?UTF-8?q?=E9=9A=A3=E3=82=8A=E5=90=88=E3=81=86=E9=A7=85=E3=81=AE=E3=81=82?= =?UTF-8?q?=E3=81=84=E3=81=A0=E3=81=AE=E7=B7=9A=E8=B7=AF=E3=81=AE=E9=95=B7?= =?UTF-8?q?=E3=81=95=E3=82=92=E8=BF=94=E3=81=99=E3=82=88=E3=81=86=E3=81=AB?= =?UTF-8?q?=E3=81=97=E3=81=9F=20(#1708)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- AGENTS.md | 2 +- docs/architecture.md | 7 +- stationapi/src/use_case/interactor/query.rs | 106 +++++++++++++++++--- 3 files changed, 100 insertions(+), 15 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 7f9b94c5..7ab2a298 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -77,7 +77,7 @@ The Worker is the workspace root package. `stationapi`, `preprocessor`, and `dat - **GTFS bus integration** – `preprocessor/src/gtfs/` reads the GTFS feeds into an in-memory representation and then projects them onto the shared `stations` / `lines` / `types` / `station_station_types` tables (`gtfs/integrate.rs`). Only routes, stops, trips, and stop_times are read; calendar, shapes, feed_info, and agencies do not affect the output. Every configured GTFS feed is imported, including Seibu Bus and Keio Bus (both downloaded from ODPT with `ODPT_ACCESS_TOKEN`). Tokyu Bus ordinary-route `BusroutePattern`, `BusstopPole`, and `BusTimetable` JSON are converted into the same representation; pattern IDs become `shape_id` values so route variants remain queryable as bus TrainTypes. The Tokyu-operated Ota, Shinagawa, and Meguro community buses use their official GTFS feeds and matching JSON routes are excluded to prevent duplicates. `ODPT_ACCESS_TOKEN` is required for authenticated sources; without it those feeds are skipped with a warning rather than failing the build. Stops whose Tokyu JSON records omit coordinates remain available to name and route queries but not coordinate searches. `transport_type` (0: rail, 1: bus) on both `stations` and `lines` keeps rail and bus records queryable side by side. GTFS IDs are namespaced per feed before import to avoid cross-operator collisions. `line_cd` (100,000,000+), `station_cd` / `station_g_cd` (200,000,000+), and bus `type_cd` / `line_group_cd` (100,000,000+) are all deterministic fnv1a hashes that stay clear of the rail data ranges. Disable the entire bus pipeline with `DISABLE_BUS_FEATURE=true`. - **Bus stop translations (readings & English)** – GTFS-JP `translations.txt` layouts differ per feed, so `load_translations` (`preprocessor/src/gtfs/parse.rs`) resolves columns by header name (Seibu ships 6 columns without `record_sub_id`; Keio and the Tokyu community feeds ship 7) and indexes each `stop_name` translation under both keys it may use: `record_id` (== the stop_id, Seibu — with the "-NN" pole suffix also mapped to the parent stop_id) and `field_value` (== the Japanese stop_name, Keio / Tokyu community, where `record_id` is left empty). `load_stops` then looks a stop's translation up by stop_id first, then by name. Keying only by `record_id` would silently drop every field_value-keyed feed, leaving `station_name_k` filled with the kanji stop_name and `station_name_r` empty. Readings arriving as half-width katakana (`ニシハチオウジ`, Keio / Tokyu community) are folded to full-width via `romaji::to_fullwidth_katakana()` before storage. - **Bus English-name fallback** – When a feed provides no English (`en`) translation for a stop — e.g. Tokyu Bus ordinary-route JSON, which carries only `dc:title` and `odpt:kana` — `stationapi/src/domain/romaji.rs::romaji_display_name()` derives a modified-Hepburn romanization (with macrons for long vowels, matching the curated rail style: Tōkyō / Kyōto / Shin-Ōsaka) from the kana reading, and the GTFS reader fills `stop_name_r` with it. The fallback never overwrites a real `en` value, and a reading with no convertible kana stays `NULL` rather than emitting a partial transcription. Because `stop_name_r` is the single upstream source that fans out into the `stations` projection, `search_by_name`, and the romanized bus route/headsign names, this supplements every English-facing surface at once. When projecting into `stations`, `station_name_rn` is filled with the plain-ASCII spelling via `romaji::strip_macrons()` (Tōkyō → Tokyo), mirroring the rail dataset's `_r` (macron) / `_rn` (macron-free) column pair. -- **Track distances** – `Station.trackDistanceFromPrevious` is the track length in meters from the station before it in the returned list, for clients that total a ride's distance (TrainLCD/MobileApp's ride log). It is distinct from `trainRoute`'s `distanceFromPrevious`, which stays the straight-line distance the app's running simulation relies on. Only `lineStations`, `lineGroupStations`, and `trainRoute` fill it (`attach_track_distances` in `QueryInteractor`, called once the order is final); the first station, the first station of each `trainRoute` leg, sections without N02 geometry, and every other query return `null`, and clients fall back to the straight line there. `preprocessor/src/track/` builds an undirected graph from N02's `RailroadSection` LineStrings, collects every pair adjacent in a line's `(e_sort, station_cd)` order or a line group's `station_station_types.id` order (with and without closed stations, plus the seam of loop services within 3 km), snaps each station to every track within 500 m, and takes the Dijkstra path minimizing `2 × snap offset + track length` — snapping to the nearest track alone picks another line of the same operator at large stations (Honmachi) and detours through a transfer station. It first measures on the operators of both stations' companies (`OPERATOR_ALIASES` maps companies whose name differs from N02's), then retries on every operator for lines running on another company's track; a path longer than max(3 × straight, straight + 5 km) counts as unmeasured, and a result shorter than the straight line is raised to it. Pairs in one station group are 0. `build.rs` writes the sorted pairs to `connections.bin`, and `index::track_distance` binary-searches it, so no index is built at isolate start. When bumping the N02 edition, change the URL and cache directory in `preprocessor/src/track/mod.rs` and the `Cache N02 railway data` step in `.github/actions/build-worker/action.yml` together and compare the unmeasured count in the `preprocessor` log. `docs/architecture.md` (駅間の線路の長さ) has the details. +- **Track distances** – `Station.trackDistanceFromPrevious` is the track length in meters from the station before it in the returned list, for clients that total a ride's distance (TrainLCD/MobileApp's ride log). It is distinct from `trainRoute`'s `distanceFromPrevious`, which stays the straight-line distance the app's running simulation relies on. Only `lineStations`, `lineGroupStations`, `trainRoute`, and `stations(ids)` fill it (`attach_track_distances` in `QueryInteractor`, called once the order is final). `stations(ids)` returns the stations in the order of `ids` — the repository keeps that order and enrichment never reorders — so a client passing a route's station IDs (MobileApp's `sids` deep links) gets the lengths between IDs adjacent in that order; a pair that is not adjacent in the track data (an ID list skipping stations) is `null`. The first station, the first station of each `trainRoute` leg, sections without N02 geometry, and every other query return `null`, and clients fall back to the straight line there. `preprocessor/src/track/` builds an undirected graph from N02's `RailroadSection` LineStrings, collects every pair adjacent in a line's `(e_sort, station_cd)` order or a line group's `station_station_types.id` order (with and without closed stations, plus the seam of loop services within 3 km), snaps each station to every track within 500 m, and takes the Dijkstra path minimizing `2 × snap offset + track length` — snapping to the nearest track alone picks another line of the same operator at large stations (Honmachi) and detours through a transfer station. It first measures on the operators of both stations' companies (`OPERATOR_ALIASES` maps companies whose name differs from N02's), then retries on every operator for lines running on another company's track; a path longer than max(3 × straight, straight + 5 km) counts as unmeasured, and a result shorter than the straight line is raised to it. Pairs in one station group are 0. `build.rs` writes the sorted pairs to `connections.bin`, and `index::track_distance` binary-searches it, so no index is built at isolate start. When bumping the N02 edition, change the URL and cache directory in `preprocessor/src/track/mod.rs` and the `Cache N02 railway data` step in `.github/actions/build-worker/action.yml` together and compare the unmeasured count in the `preprocessor` log. `docs/architecture.md` (駅間の線路の長さ) has the details. - **TTS metadata** – `Station`, `StationNested`, `Line`, `LineNested`, `TrainType`, and `TrainTypeNested` expose `name_ipa` / `name_roman_ipa` plus `name_tts_segments` for multi-segment pronunciation output. Use `name_tts_segments` when clients need per-token SSML construction for mixed-language names such as `Kasai-Rinkai Park`. - **Connected routes** – `connectedRoutes` finds transfer routes automatically, like a journey planner, using a frequency-based RAPTOR search in `stationapi/src/domain/route_search.rs`. Each rail line group is a pattern, station groups are the transfer nodes, and ride times come from `arrival_estimation`; bus lines are excluded. The cost adds a per-boarding wait by `TrainTypeKind` (limited express 15 min, express / high-speed rapid 5 min, others 3 min) and a 3-minute transfer walk — without the wait, infrequent limited expresses would beat the Yamanote Line. Rounds give the time/transfer Pareto set; alternatives come from re-searching with one leg's parallel line groups banned along that leg (at most 8 searches), and are dropped beyond 1.15 × best + 15 min or with two more transfers than the Pareto set. Alternative routes that stop at the same station group in two different legs (backtracking to re-board a banned train) are dropped; pass-through stations are not counted, and the Pareto routes of the first search are never dropped this way (otherwise a station `stationsByName` reports as reachable could get no route). Results are ranked by cost + 5 min per transfer and capped at 6. The time and transfer count stay internal (the API does not return them), so ordering happens on the server: `sortBy: ConnectedRouteSort` picks `Recommended` (the ranking above, the default when omitted), `ArrivalTime` (the estimated time `estimateArrivalTimes` also reports — excluding the first train's wait — then fewer transfers), or `TransferCount` (fewer transfers, then earlier arrival). `route_search::sort_journeys` only reorders the set `search` returned, with a stable sort so ties keep the recommended order; the set itself never depends on `sortBy`. The network (every rail line group plus its time estimates, about 190 ms natively) is built lazily into a `OnceLock` by `StationRepository::get_route_network` on the first `connectedRoutes` call, so other queries never pay for it. Each route is a list of `legs` shaped for the app's one-train-at-a-time flow: every leg carries its boarding and alighting `Station`, both on the line of the line group the search rode — so at a transfer the previous leg's alighting station and the next leg's boarding station may be different stations of one station group — `stationGroupIds`, the station groups from boarding to alighting in travel order including pass-through stations (the search's `JourneyLeg.station_group_ids` as-is — station groups rather than station IDs so the client can match them against whichever train type it picks, which may run on another line; a group appears twice on patterns such as the Oedo Line's Tochomae), and `trainTypes`, every train type usable on that leg (real `groupId`s, so the client picks one and calls `lineGroupStations`): it is exactly `routeTypes(boarding station group, alighting station group, alighting station's line)` (same dedup, same `lines`, same order — the use case calls `get_train_types`), because the search collapses parallel services such as local and rapid into one route and the app needs them to list types and default to the local. `viaLineId`, like `routeTypes`, is the line of the tapped search result and keeps only routes whose last leg arrives on that line. `estimateArrivalTimes` and `trainRoute` accept `legs: [RouteLegInput!]` (the `groupId` of the train type picked from each leg's `trainTypes`, plus the leg's `fromStation.id` and `toStation.id`) and then return values for the whole transfer route. A leg endpoint missing from the chosen line group is matched by station group (the picked local may stop at another line's station of the same group), preferring an exact `station_cd` and, among same-group candidates, the pair giving the shortest slice (through services list two stations of a junction group). ETA estimates each leg on its own line group only and chains them from the origin, adding the 3-minute walk and the next train type's wait at each transfer (the same allowance the search ranks by), returning one route with an empty `id`; `trainRoute` concatenates each leg's segments (each leg restarts at distance 0). Both slice legs with the same function, taking the shorter arc on loop lines, so their station sequences match. More than `MAX_RIDES` (6) legs — more than `connectedRoutes` ever returns — legs that do not connect, ends that differ from `fromStationId` / `toStationId`, or combining `legs` with `viaLineIds` / `directionId` / `lineGroupId` are errors. `docs/architecture.md` (乗換経路探索) has the details, and `docs/route-search.md` documents the search internals (data structures, the scan, pruning, alternatives, determinism). - Changes to the published contract require coordinated updates to `schema/public.graphql`, the async-graphql types in `src/graphql/`, and, when the shape of a value changes, `stationapi/src/model.rs` and the DTO conversions. diff --git a/docs/architecture.md b/docs/architecture.md index 5f2f56dc..73afaaa2 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -222,8 +222,11 @@ preprocessor の実行は 1 秒ほどです。 Worker では `build.rs` が 1 行 = `i32` 3 つの固定長バイナリ (`connections.bin`、 約 130KB) に変換し、組の昇順に並べます。ランタイムは索引を作らずに二分探索で 引くので、コールドスタートの費用は増えません。値を埋めるのは、並びを返す -`lineStations`・`lineGroupStations`・`trainRoute` だけです (他の問い合わせでは -直前の駅が無いので `null`)。`trainRoute` を `legs` で呼んだ場合は、各区間の +`lineStations`・`lineGroupStations`・`trainRoute`・`stations(ids)` だけです +(他の問い合わせでは直前の駅が無いので `null`)。`stations(ids)` は駅を指定した +ID の順に返すので、経路の駅を順に渡すクライアント (MobileApp の `sids` 形式の +ディープリンク) は、その順で隣り合う組の長さを受け取れます。途中の駅を省いた +並びなど、線路のデータ上で隣り合わない組は `null` です。`trainRoute` を `legs` で呼んだ場合は、各区間の 先頭の駅も `null` になります。 N02 は `data/N02-25/` にキャッシュします (git 管理外。CI では `actions/cache` で diff --git a/stationapi/src/use_case/interactor/query.rs b/stationapi/src/use_case/interactor/query.rs index a14a5e2b..85c78bdd 100644 --- a/stationapi/src/use_case/interactor/query.rs +++ b/stationapi/src/use_case/interactor/query.rs @@ -90,15 +90,13 @@ where station_ids: &[u32], transport_type: TransportTypeFilter, ) -> Result, UseCaseError> { - let stations = self.station_repository.get_by_id_vec(station_ids).await?; - // Filter by transport_type - let stations: Vec = stations - .into_iter() - .filter(|s| matches_transport_filter(s.transport_type, transport_type)) - .collect(); - let stations = self - .update_station_vec_with_attributes(stations, None, transport_type, true) + let mut stations = self + .enriched_stations_by_id_vec(station_ids, transport_type) .await?; + // 並びは指定された ID の順 (repository がそう返し、付帯情報の付与も + // 並びを変えない)。ディープリンクで経路の駅を ID の順に渡すクライアントは、 + // その順で隣り合う組の線路の長さを乗車距離に使う + self.attach_track_distances(&mut stations).await?; Ok(stations) } @@ -1076,9 +1074,10 @@ where .into_iter() .collect(); - // 乗降駅は stations クエリと同じ付帯情報を付ける + // 乗降駅は stations クエリと同じ付帯情報を付ける。ID の並びは HashSet 由来で + // 経路の順ではないため、線路の長さは付けない let stations: HashMap = self - .get_stations_by_id_vec(&station_ids, TransportTypeFilter::Rail) + .enriched_stations_by_id_vec(&station_ids, TransportTypeFilter::Rail) .await? .into_iter() .map(|station| (station.station_cd, station)) @@ -1237,6 +1236,24 @@ where Ok(group_of) } + /// 指定された ID の駅に、stations クエリと同じ付帯情報を付けて返す。並びは ID の順。 + /// 線路の長さは付けない (ID の並びが経路の順とは限らない呼び出し元があるため)。 + async fn enriched_stations_by_id_vec( + &self, + station_ids: &[u32], + transport_type: TransportTypeFilter, + ) -> Result, UseCaseError> { + let stations: Vec = self + .station_repository + .get_by_id_vec(station_ids) + .await? + .into_iter() + .filter(|s| matches_transport_filter(s.transport_type, transport_type)) + .collect(); + self.update_station_vec_with_attributes(stations, None, transport_type, true) + .await + } + /// 返す駅の並びで隣り合う 2 駅のあいだの線路の長さを、後ろの駅の /// `track_distance_from_previous` に入れる。先頭の駅は `None` のまま。 /// @@ -3656,6 +3673,8 @@ mod tests { stations_by_group: Vec, bus_stops: Vec, stations_by_line_group: Vec, + /// get_track_distances がどの組にも返す長さ + track_distance: Option, } impl ConfigurableMockStationRepository { @@ -3664,6 +3683,7 @@ mod tests { stations_by_group, bus_stops, stations_by_line_group: vec![], + track_distance: None, } } @@ -3671,6 +3691,11 @@ mod tests { self.stations_by_line_group = stations; self } + + fn with_track_distance(mut self, distance: f64) -> Self { + self.track_distance = Some(distance); + self + } } #[async_trait::async_trait] @@ -3690,6 +3715,12 @@ mod tests { &EstimationParams::default(), ))) } + async fn get_track_distances( + &self, + pairs: &[(u32, u32)], + ) -> Result>, DomainError> { + Ok(vec![self.track_distance; pairs.len()]) + } async fn find_by_id(&self, _: u32) -> Result, DomainError> { Ok(None) } @@ -5044,6 +5075,25 @@ mod tests { .collect() } + /// 乗降駅はまとめて引くが、その並びは経路の順ではない。どの組にも線路の + /// 長さがあるとしても、乗降駅には付けない + #[tokio::test] + async fn test_get_connected_routes_leaves_track_distances_unset() { + let mut interactor = create_connected_route_interactor(); + interactor.station_repository = interactor.station_repository.with_track_distance(1.0); + + let routes = interactor + .get_connected_routes(1, 4, None, JourneySort::Recommended) + .await + .unwrap(); + + assert!(!routes.is_empty()); + for leg in routes.iter().flat_map(|route| route.legs.iter()) { + assert_eq!(leg.from_station.track_distance_from_previous, None); + assert_eq!(leg.to_station.track_distance_from_previous, None); + } + } + #[tokio::test] async fn test_get_connected_routes_returns_legs_on_real_line_groups() { let interactor = create_connected_route_interactor(); @@ -5756,8 +5806,17 @@ mod tests { async fn find_by_id(&self, _: u32) -> Result, DomainError> { Ok(None) } - async fn get_by_id_vec(&self, _: &[u32]) -> Result, DomainError> { - Ok(vec![]) + /// 本物と同じく、指定された ID の順に返す + async fn get_by_id_vec(&self, ids: &[u32]) -> Result, DomainError> { + Ok(ids + .iter() + .filter_map(|&id| { + self.line_group_stations + .iter() + .find(|s| s.station_cd as u32 == id) + .cloned() + }) + .collect()) } async fn get_by_line_id( &self, @@ -6110,6 +6169,29 @@ mod tests { assert_eq!(distances, vec![None, Some(100.0), Some(101.0)]); } + /// stations(ids) は指定した ID の順で隣り合う組の長さを返す。逆向きの並びも + /// その向きで引き、途中の駅を省いた組とデータの無い組は None + #[tokio::test] + async fn stations_by_ids_carry_track_distances_in_the_requested_order() { + let (interactor, _) = build_interactor(build_line_group(7)); + + let stations = interactor + .get_stations_by_id_vec( + &[1003, 1002, 1001, 1005, 1006], + TransportTypeFilter::RailAndBus, + ) + .await + .unwrap(); + + let ids: Vec = stations.iter().map(|s| s.station_cd).collect(); + assert_eq!(ids, vec![1003, 1002, 1001, 1005, 1006]); + let distances: Vec> = stations + .iter() + .map(|s| s.track_distance_from_previous) + .collect(); + assert_eq!(distances, vec![None, Some(102.0), Some(101.0), None, None]); + } + /// 逆向きに切り出した区間では、進む向きで直前の駅からの長さになる #[tokio::test] async fn train_route_track_distances_follow_the_travel_direction() {