Repository navigation
Expand file tree
/
Copy pathbuild.rs
More file actions
212 lines (197 loc) · 9.71 KB
/
Copy pathbuild.rs
File metadata and controls
212 lines (197 loc) · 9.71 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
//! station_station_types.csv と connections.csv を固定長バイナリへ事前変換する。
//!
//! station_station_types は 41,250 行あり、isolate 起動時の CSV パースが
//! コールドスタートの大半を占める。全列が整数なので 1 行 = i32 x 4 の固定長に
//! しておけば、ランタイムではスライスを読むだけで済む。
//!
//! connections (隣り合う駅のあいだの線路の長さ) は駅の組で並べた 1 行 = i32 x 3 に
//! しておき、ランタイムは索引を作らずに二分探索で引く。
use std::{env, fs, path::Path, path::PathBuf};
/// CSV に NULL 表現が無いため、欠損値を i32::MIN で表す
const NULL_I32: i32 = i32::MIN;
/// Workers が読む CSV を OUT_DIR へ集める。
///
/// generated/ があればそれを使う。無ければ data/*.csv にフォールバックする。
/// generated は preprocessor が組み立てたもので、data/*.csv には無い
/// 各駅停車の生成系統や、GTFS 由来のバス停・バス路線を含む。
fn stage_csv(out_dir: &Path, name: &str, fallback: &str) -> bool {
let generated = Path::new("generated").join(name);
println!("cargo:rerun-if-changed=generated/{name}");
println!("cargo:rerun-if-changed={fallback}");
let (src, from_generated) = if generated.is_file() {
(generated, true)
} else {
(PathBuf::from(fallback), false)
};
fs::copy(&src, out_dir.join(name))
.unwrap_or_else(|e| panic!("{} を配置できない: {e}", src.display()));
from_generated
}
fn main() {
// 本番と同じデータを使うには preprocessor の出力が要る。preprocessor は
// 列車種別を持たない路線へ各駅停車の系統を補う (約2,400行)。この行は
// data/*.csv に存在しないため、CSV を直接読むと 2,268 駅の
// has_train_types が false になり本番と挙動がずれる。
//
// CI では `cargo run --profile tool -p stationapi-preprocessor` で generated/ を
// 作り、それをここで優先して読む。
let generated = Path::new("generated/station_station_types.csv");
let fallback = Path::new("data/5!station_station_types.csv");
let (csv_path, has_generated) = if generated.is_file() {
(generated, true)
} else {
println!(
"cargo:warning=generated/station_station_types.csv が無いため \
data/5!station_station_types.csv を使用します。生成される各駅停車の系統が \
含まれないため、本番と挙動が異なります"
);
(fallback, false)
};
let csv_path = csv_path.to_string_lossy().to_string();
println!("cargo:rerun-if-changed={csv_path}");
println!("cargo:rerun-if-changed=generated/station_station_types.csv");
let out_dir = PathBuf::from(env::var("OUT_DIR").expect("OUT_DIR"));
let staged = [
stage_csv(&out_dir, "stations.csv", "data/3!stations.csv"),
stage_csv(&out_dir, "lines.csv", "data/2!lines.csv"),
stage_csv(&out_dir, "companies.csv", "data/1!companies.csv"),
stage_csv(&out_dir, "types.csv", "data/4!types.csv"),
stage_csv(&out_dir, "aliases.csv", "data/6!aliases.csv"),
stage_csv(&out_dir, "line_aliases.csv", "data/7!line_aliases.csv"),
stage_csv(&out_dir, "connections.csv", "data/8!connections.csv"),
// station_station_types は下の sst 変換でも参照するが、
// 混在判定に含めるためここでも存在を見る
Path::new("generated/station_station_types.csv").is_file(),
];
// 混在を許すと、例えば generated の station_station_types だけが入り
// stations が data/*.csv のままになる。生成された系統の station_cd が
// 駅索引に存在せず、列車種別が欠落した Worker が出来上がってしまう。
// cargo:warning は CI で見落とされるため、混在はビルドを止める。
let generated_count = staged.iter().filter(|v| **v).count();
if generated_count != 0 && generated_count != staged.len() {
panic!(
"generated が {} / {} ファイルしかありません。\n\
生成物と data/*.csv が混ざるとデータが不整合になります。\n\
`cargo run --profile tool -p stationapi-preprocessor` で全テーブルを書き出すか、\n\
generated を削除して data/*.csv だけを使ってください。",
generated_count,
staged.len()
);
}
// どちらのデータを埋め込んだか。所要時間のベンチマーク (src/travel_times.rs) の
// 記録は生成データで作るので、data/*.csv のときは記録と比べない
println!(
"cargo:rustc-env=STATIONAPI_EMBEDDED_DATA={}",
if generated_count == 0 {
"data"
} else {
"generated"
}
);
if generated_count == 0 {
println!(
"cargo:warning=generated が無いため data/*.csv を使用します。\
生成される各駅停車の系統や GTFS 由来のバスデータが含まれないため、\
本番と挙動が異なります"
);
}
let mut reader = csv::ReaderBuilder::new()
.has_headers(true)
.from_path(&csv_path)
.expect("station_station_types.csv を開けない");
let headers = reader.headers().expect("ヘッダを読めない").clone();
let col = |name: &str| headers.iter().position(|h| h.trim() == name);
let (Some(i_station), Some(i_type)) = (col("station_cd"), col("type_cd")) else {
panic!("station_cd / type_cd 列が見つからない");
};
let i_group = col("line_group_cd");
let i_pass = col("pass");
// 生成物は SERIAL 採番後の実 id を持つ。CSV は "DEFAULT" なので行順で採番する。
let i_id = if has_generated { col("id") } else { None };
let num = |r: &csv::StringRecord, i: Option<usize>| -> i32 {
i.and_then(|i| r.get(i))
.and_then(|v| v.trim().parse::<i32>().ok())
.unwrap_or(NULL_I32)
};
let mut out: Vec<u8> = Vec::with_capacity(45_000 * 16);
let mut expected_id = 0i32;
for record in reader.records().flatten() {
let station_cd = num(&record, Some(i_station));
let type_cd = num(&record, Some(i_type));
// ランタイム側の CSV パースと同じく、必須列が壊れている行は落とす
if station_cd == NULL_I32 || type_cd == NULL_I32 {
continue;
}
// ランタイムは行順で id を振り直すため、生成物の id が連番でなければ
// 停車順序がずれる。ずれていたらビルドを止める。
expected_id += 1;
if let Some(i) = i_id {
let actual = num(&record, Some(i));
assert_eq!(
actual, expected_id,
"station_station_types.id が連番ではない (期待 {expected_id}, 実際 {actual})"
);
}
for value in [
station_cd,
type_cd,
num(&record, i_group),
num(&record, i_pass),
] {
out.extend_from_slice(&value.to_le_bytes());
}
}
fs::write(out_dir.join("sst.bin"), &out).expect("sst.bin を書けない");
println!("cargo:warning=sst.bin: {} 行", out.len() / 16);
write_connections(&out_dir);
}
/// connections.csv を (station_cd1, station_cd2, 整数メートル) の固定長バイナリへ
/// 変換する。組は小さい station_cd を先にして昇順に並べる (ランタイムの二分探索用)。
///
/// data/8!connections.csv へフォールバックした場合は手入力の行だけになる。
/// 並びや値の書式が生成物と違ってもよいよう、ここで正規化する。
fn write_connections(out_dir: &Path) {
let mut reader = csv::ReaderBuilder::new()
.has_headers(true)
.from_path(out_dir.join("connections.csv"))
.expect("connections.csv を開けない");
let headers = reader.headers().expect("ヘッダを読めない").clone();
let col = |name: &str| {
headers
.iter()
.position(|h| h.trim() == name)
.unwrap_or_else(|| panic!("connections.csv に {name} 列が無い"))
};
let (i_a, i_b, i_distance) = (col("station_cd1"), col("station_cd2"), col("distance"));
let mut rows: Vec<(i32, i32, i32)> = Vec::new();
for record in reader.records() {
let record = record.expect("connections.csv の行を読めない");
let parse = |i: usize| record.get(i).map(str::trim).unwrap_or("");
let (Ok(a), Ok(b)) = (parse(i_a).parse::<i32>(), parse(i_b).parse::<i32>()) else {
panic!("connections.csv の駅コードが整数ではない: {record:?}");
};
let distance = parse(i_distance)
.parse::<f64>()
.ok()
.filter(|d| d.is_finite() && *d >= 0.0 && *d <= i32::MAX as f64)
.unwrap_or_else(|| panic!("connections.csv の distance が不正: {record:?}"));
rows.push((a.min(b), a.max(b), distance.round() as i32));
}
rows.sort_unstable();
for w in rows.windows(2) {
assert!(
(w[0].0, w[0].1) != (w[1].0, w[1].1),
"connections.csv に同じ駅の組が 2 行ある: {} - {}",
w[0].0,
w[0].1
);
}
let mut out: Vec<u8> = Vec::with_capacity(rows.len() * 12);
for (a, b, distance) in &rows {
for value in [*a, *b, *distance] {
out.extend_from_slice(&value.to_le_bytes());
}
}
fs::write(out_dir.join("connections.bin"), &out).expect("connections.bin を書けない");
println!("cargo:warning=connections.bin: {} 行", rows.len());
}