Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 14 additions & 2 deletions .github/actions/build-worker/action.yml
Original file line number Diff line number Diff line change
Expand Up @@ -47,9 +47,21 @@ runs:
target
key: worker-${{ runner.os }}-${{ hashFiles('**/Cargo.lock') }}

# N02 の取得先は国の配信サーバーで、取得に失敗すると preprocessor が失敗する。
# サーバーの一時的な不調で検証やデプロイが止まらないよう、展開した GeoJSON を
# キャッシュする。中身は版で決まるので、キーは版だけにする
# (preprocessor/src/track/mod.rs の CACHE_DIR と揃えること)。
- name: Cache N02 railway data
uses: actions/cache@v4
with:
path: data/N02-25
key: n02-25

# data/*.csv をそのまま Worker へ渡すと本番と挙動が変わる。列車種別を
# 持たない路線には各駅停車の系統を補う必要があり (約2,400行)、バス停と
# バス路線は GTFS / ODPT から起こす必要があるため。
# バス路線は GTFS / ODPT から、駅間の線路の長さは国土数値情報 (N02) から
# 起こす必要があるため。N02 はトークン不要で、取得に失敗すると preprocessor
# 自体が失敗する。
- name: Build generated data
shell: bash
env:
Expand Down Expand Up @@ -82,7 +94,7 @@ runs:
- name: Verify generated data
shell: bash
run: |
for t in companies lines stations types station_station_types aliases line_aliases; do
for t in companies lines stations types station_station_types aliases line_aliases connections; do
test -s "generated/$t.csv" || { echo "::error::$t.csv が空"; exit 1; }
echo "$t: $(python3 -c "import csv,sys;print(sum(1 for _ in csv.reader(open(sys.argv[1]))) - 1)" "generated/$t.csv") rows"
done
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@ data/TokyuBus-ShinagawaCity-GTFS/
data/TokyuBus-MeguroCity-GTFS/
data/TokyuBus-ODPT/
data/KeioBus-GTFS/
data/N02-25/
scripts/.osm_cache/
scripts/.gtfs_cache/
__pycache__/
Expand Down
11 changes: 7 additions & 4 deletions AGENTS.md

Large diffs are not rendered by default.

3 changes: 2 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@ help:
@echo " check - Type-check every crate (worker targets wasm32)"
@echo " fmt - Check formatting"
@echo " clippy - Lint every crate"
@echo " data - Rebuild generated/*.csv from data/ and the GTFS feeds"
@echo " data - Rebuild generated/*.csv from data/, the GTFS feeds, and the MLIT railway data"
@echo " build - Build the Worker (wasm)"
@echo " dev - Run the Worker locally (wrangler dev)"
@echo " deploy - Deploy to staging (dev branch only)"
Expand Down Expand Up @@ -46,6 +46,7 @@ clippy:
cargo clippy --target wasm32-unknown-unknown -p stationapi-worker --all-targets -- -D warnings

# Worker が読むデータを作り直す。data/*.csv や GTFS が変わったら実行する。
# 駅間の線路の長さに使う国土数値情報 (N02) は data/N02-25/ にキャッシュする。
data:
cargo run --profile tool -p stationapi-preprocessor

Expand Down
12 changes: 11 additions & 1 deletion README.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,16 @@ A GraphQL API that provides nearby Japanese train stations and bus stops, runnin

This project includes a comprehensive dataset of Japanese railway information in the `data/` directory. The data is maintained in CSV format and contributions are primarily targeted at Japanese speakers. For detailed information about data structure and contribution guidelines, please refer to [data/README.md](data/README.md).

## Data Sources

- Track lengths between adjacent stations (`Station.trackDistanceFromPrevious`) are
derived from the MLIT National Land Numerical Information railway data (N02),
licensed under CC BY 4.0:
「国土数値情報(鉄道データ)」(国土交通省)
(https://nlftp.mlit.go.jp/ksj/gml/datalist/KsjTmplt-N02-2025.html) を加工して作成
- Bus stops and routes are derived from the GTFS and ODPT feeds listed in
`preprocessor/src/gtfs/feed.rs` and `preprocessor/src/gtfs/odpt.rs`.

## Contributors ✨

Thanks goes to these wonderful people ([emoji key](https://allcontributors.org/docs/en/emoji-key)):
Expand Down Expand Up @@ -62,7 +72,7 @@ into the WASM binary at build time.
rustup target add wasm32-unknown-unknown
cargo install worker-build --locked

make data # build generated/*.csv from data/ and the GTFS feeds
make data # build generated/*.csv from data/, the GTFS feeds, and the MLIT railway data
make build # build the Worker (wasm)
make dev # run it locally on http://127.0.0.1:8787
```
Expand Down
67 changes: 63 additions & 4 deletions build.rs
Original file line number Diff line number Diff line change
@@ -1,8 +1,11 @@
//! station_station_types.csv を固定長バイナリへ事前変換する。
//! station_station_types.csv と connections.csv を固定長バイナリへ事前変換する。
//!
//! この CSV は 41,250 行あり、isolate 起動時の CSV パースがコールドスタートの
//! 大半を占める。全列が整数なので 1 行 = i32 x 4 の固定長にしておけば、
//! ランタイムではスライスを読むだけで済む。
//! station_station_types は 41,250 行あり、isolate 起動時の CSV パースが
//! コールドスタートの大半を占める。全列が整数なので 1 行 = i32 x 4 の固定長に
//! しておけば、ランタイムではスライスを読むだけで済む。
//!
//! connections (隣り合う駅のあいだの線路の長さ) は駅の組で並べた 1 行 = i32 x 3 に
//! しておき、ランタイムは索引を作らずに二分探索で引く。

use std::{env, fs, path::Path, path::PathBuf};

Expand Down Expand Up @@ -62,6 +65,7 @@ fn main() {
stage_csv(&out_dir, "types.csv", "data/4!types.csv"),
stage_csv(&out_dir, "aliases.csv", "data/6!aliases.csv"),
stage_csv(&out_dir, "line_aliases.csv", "data/7!line_aliases.csv"),
stage_csv(&out_dir, "connections.csv", "data/8!connections.csv"),
// station_station_types は下の sst 変換でも参照するが、
// 混在判定に含めるためここでも存在を見る
Path::new("generated/station_station_types.csv").is_file(),
Expand Down Expand Up @@ -140,4 +144,59 @@ fn main() {

fs::write(out_dir.join("sst.bin"), &out).expect("sst.bin を書けない");
println!("cargo:warning=sst.bin: {} 行", out.len() / 16);

write_connections(&out_dir);
}

/// connections.csv を (station_cd1, station_cd2, 整数メートル) の固定長バイナリへ
/// 変換する。組は小さい station_cd を先にして昇順に並べる (ランタイムの二分探索用)。
///
/// data/8!connections.csv へフォールバックした場合は手入力の行だけになる。
/// 並びや値の書式が生成物と違ってもよいよう、ここで正規化する。
fn write_connections(out_dir: &Path) {
let mut reader = csv::ReaderBuilder::new()
.has_headers(true)
.from_path(out_dir.join("connections.csv"))
.expect("connections.csv を開けない");
let headers = reader.headers().expect("ヘッダを読めない").clone();
let col = |name: &str| {
headers
.iter()
.position(|h| h.trim() == name)
.unwrap_or_else(|| panic!("connections.csv に {name} 列が無い"))
};
let (i_a, i_b, i_distance) = (col("station_cd1"), col("station_cd2"), col("distance"));

let mut rows: Vec<(i32, i32, i32)> = Vec::new();
for record in reader.records() {
let record = record.expect("connections.csv の行を読めない");
let parse = |i: usize| record.get(i).map(str::trim).unwrap_or("");
let (Ok(a), Ok(b)) = (parse(i_a).parse::<i32>(), parse(i_b).parse::<i32>()) else {
panic!("connections.csv の駅コードが整数ではない: {record:?}");
};
let distance = parse(i_distance)
.parse::<f64>()
.ok()
.filter(|d| d.is_finite() && *d >= 0.0 && *d <= i32::MAX as f64)
.unwrap_or_else(|| panic!("connections.csv の distance が不正: {record:?}"));
rows.push((a.min(b), a.max(b), distance.round() as i32));
}
rows.sort_unstable();
for w in rows.windows(2) {
assert!(
(w[0].0, w[0].1) != (w[1].0, w[1].1),
"connections.csv に同じ駅の組が 2 行ある: {} - {}",
w[0].0,
w[0].1
);
}

let mut out: Vec<u8> = Vec::with_capacity(rows.len() * 12);
for (a, b, distance) in &rows {
for value in [*a, *b, *distance] {
out.extend_from_slice(&value.to_le_bytes());
}
}
fs::write(out_dir.join("connections.bin"), &out).expect("connections.bin を書けない");
println!("cargo:warning=connections.bin: {} 行", rows.len());
}
31 changes: 18 additions & 13 deletions data/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@
| `5!station_station_types.csv` | 駅と列車種別の関連情報 |
| `6!aliases.csv` | 路線の別名・愛称情報 |
| `7!line_aliases.csv` | 駅と路線別名の関連情報 |
| `8!connections.csv` | 駅間の接続・距離情報 |
| `8!connections.csv` | 駅間の線路の長さの手修正 |

## 🏢 1!companies.csv - 鉄道会社情報

Expand Down Expand Up @@ -199,26 +199,31 @@
- `station_cd`は`3!stations.csv`に存在する値を使用
- `alias_cd`は`6!aliases.csv`に存在する値を使用

## 🚇 8!connections.csv - 駅間の接続・距離情報
## 🚇 8!connections.csv - 駅間の線路の長さの手修正

> ⚠️ **注意**: このファイルは現在どこでも使用されていません。将来的に経路計算機能で使用される予定ですが、実装時期は未定です。
隣り合う駅のあいだの線路の長さ(`Station.trackDistanceFromPrevious`)は、
preprocessor が国土数値情報の鉄道データ(N02)から自動で求めます。このファイルは、
その値が実際と合わない区間を手で直すためのものです。ここに書いた値は計算結果より
優先されます。計算の方法は [docs/architecture.md の「駅間の線路の長さ」](../docs/architecture.md#駅間の線路の長さ)
を参照してください。

### フィールド説明

| フィールド名 | 型 | 必須 | 説明 | 例 |
| ------------- | ---- | ---- | -------------------- | ------------- |
| `id` | 数値 | ✓ | 主キー(設計未確定) | `1` |
| `station_cd1` | 数値 | ✓ | 起点駅コード | `100201` |
| `station_cd2` | 数値 | ✓ | 終点駅コード | `100202` |
| `distance` | 数値 | - | 駅間距離(メートル) | `6140.152858` |
| フィールド名 | 型 | 必須 | 説明 | 例 |
| ------------- | ---- | ---- | ------------------------------ | --------- |
| `id` | 数値 | ✓ | 行番号(`DEFAULT` でもよい) | `1` |
| `station_cd1` | 数値 | ✓ | 駅コード | `100201` |
| `station_cd2` | 数値 | ✓ | 駅コード(`station_cd1` の隣) | `100202` |
| `distance` | 数値 | ✓ | 線路の長さ(メートル) | `6800` |

### 入力時の注意点

- 駅コードは`3!stations.csv`に存在する値を使用
- 距離はメートル単位で入力
- 方向性がある場合は、両方向のレコードを作成
- **現在は使用されていないため、データ入力の優先度は低い**
- **将来的な実装時に仕様が変更される可能性がある**
- 2 駅は、`lineStations` / `lineGroupStations` が返す並びで隣り合う駅にする(それ以外の組は API で使われない)
- 組の向きは問わない。同じ組(向きを入れ替えたものを含む)を 2 行書かない
- 距離はメートル単位の 0 以上の数値。小数は四捨五入して整数メートルになる
- 値の出典(営業キロ・実キロなど)は PR に書く
- `cargo run -p data_validator` が上記を検査する

## 📝 共通ガイドライン

Expand Down
69 changes: 68 additions & 1 deletion data_validator/src/main.rs
Original file line number Diff line number Diff line change
Expand Up @@ -123,17 +123,26 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
println!("[INVALID] {message}");
}

let mut rdr = ReaderBuilder::new().from_path(data_path.join("8!connections.csv"))?;
let connection_records: Vec<StringRecord> = rdr.records().collect::<Result<Vec<_>, _>>()?;
let invalid_connections = validate_connections(&connection_records, &station_ids);
for message in &invalid_connections {
println!("[INVALID] {message}");
}

let has_err = !invalid_station_ids.is_empty()
|| !invalid_type_ids.is_empty()
|| !invalid_line_ids.is_empty()
|| !invalid_station_orders.is_empty();
|| !invalid_station_orders.is_empty()
|| !invalid_connections.is_empty();

if has_err {
let report = build_markdown_report(
&invalid_station_ids,
&invalid_type_ids,
&invalid_line_ids,
&invalid_station_orders,
&invalid_connections,
);
let report_path =
std::env::var("VALIDATION_REPORT_PATH").unwrap_or("/tmp/validation_report.md".into());
Expand Down Expand Up @@ -194,11 +203,57 @@ fn validate_station_orders(station_records: &[StringRecord]) -> Vec<String> {
errors
}

/// `8!connections.csv` (手で直した駅間の線路の長さ) の各行について、両駅が
/// `3!stations.csv` にあり、別の駅で、長さが 0 以上の数値で、同じ組
/// (向きを問わない) が 2 行無いことを検証する。preprocessor はこの値を
/// 国土数値情報から求めた値より優先する。
fn validate_connections(records: &[StringRecord], station_ids: &HashSet<u32>) -> Vec<String> {
const COL_STATION_CD1: usize = 1;
const COL_STATION_CD2: usize = 2;
const COL_DISTANCE: usize = 3;

let mut errors: Vec<String> = Vec::new();
let mut seen: HashSet<(u32, u32)> = HashSet::new();
for record in records {
let line = record.iter().collect::<Vec<&str>>().join(",");
let station = |i: usize| record.get(i).and_then(|v| v.trim().parse::<u32>().ok());
let (Some(a), Some(b)) = (station(COL_STATION_CD1), station(COL_STATION_CD2)) else {
errors.push(format!("8!connections.csv: 駅コードを読めません: {line}"));
continue;
};
for cd in [a, b] {
if !station_ids.contains(&cd) {
errors.push(format!(
"8!connections.csv: 存在しない station_cd {cd} を参照しています: {line}"
));
}
}
if a == b {
errors.push(format!("8!connections.csv: 同じ駅どうしの行です: {line}"));
}
let distance = record
.get(COL_DISTANCE)
.and_then(|v| v.trim().parse::<f64>().ok());
if !distance.is_some_and(|d| d.is_finite() && d >= 0.0) {
errors.push(format!(
"8!connections.csv: distance が 0 以上の数値ではありません: {line}"
));
}
if !seen.insert((a.min(b), a.max(b))) {
errors.push(format!(
"8!connections.csv: 同じ駅の組 (向きを問わない) が 2 行あります: {line}"
));
}
}
errors
}

fn build_markdown_report(
invalid_station_ids: &[String],
invalid_type_ids: &[String],
invalid_line_ids: &[String],
invalid_station_orders: &[String],
invalid_connections: &[String],
) -> String {
let mut md = String::new();

Expand Down Expand Up @@ -273,6 +328,18 @@ fn build_markdown_report(
md.push('\n');
}

if !invalid_connections.is_empty() {
md.push_str(&format!(
"### 駅間の線路の長さのエラー ({} 件)\n\n",
invalid_connections.len()
));
md.push_str("`8!connections.csv` の行が不正です。\n\n");
for message in invalid_connections {
md.push_str(&format!("- {}\n", escape_markdown_cell(message)));
}
md.push('\n');
}

md
}

Expand Down
Loading
Loading