From 1c95398a9e6d0a4c03ca569ebe49d35e00433642 Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Wed, 23 Sep 2026 14:01:06 -0400 Subject: [PATCH 1/2] lex each token as one terminal in the basic grammar A token that started like a number lexed as a NUMBER followed by a word: `START_DATE_TIME 1997-07-16T19:20:30` gave `[1997, '-07-16T19:20:30']`, and TDIS then failed to write. The same applied to any token with a leading digit, such as a file name `1model.ts`. Lark's lexer takes the first terminal that matches, not the longest match, and NUMBER sorted ahead of word. Every whitespace-delimited token now lexes as a single TOKEN terminal, and the transformer makes it a number if the whole token is one. This also reads signed numbers (`-5`) as numbers instead of strings, and Fortran `D` exponents (`1D-5`) as floats. 21 corpus models now write. 5 round-trip cleanly; the other 16 show the single-name AUXILIARY bug, and test001h_rch_array3 also drops a TAS period array. Co-Authored-By: Claude Opus 5.5 --- flopy4/mf6/codec/reader/grammar/basic.lark | 13 +++--- flopy4/mf6/codec/reader/transformer/basic.py | 17 ++++---- test/mf6/test_mf6_codec.py | 19 +++++++++ test/mf6/test_mf6_io_roundtrip.py | 42 ++++++++------------ 4 files changed, 53 insertions(+), 38 deletions(-) diff --git a/flopy4/mf6/codec/reader/grammar/basic.lark b/flopy4/mf6/codec/reader/grammar/basic.lark index 63f15f8f..c67eb7ad 100644 --- a/flopy4/mf6/codec/reader/grammar/basic.lark +++ b/flopy4/mf6/codec/reader/grammar/basic.lark @@ -1,17 +1,20 @@ start: [_NL*] (block [_NL*])* block: "begin"i block_name _NL _list "end"i block_name [_NL+] -block_name: CNAME [NUMBER] +block_name: CNAME [TOKEN] _list: line* -line: item* _NL+ -item: word | NUMBER -word: /[a-zA-Z0-9'().\/:;<=>?@_~+\[\\-]+/ +line: TOKEN* _NL+ +// Every whitespace-delimited token lexes as one terminal, and the +// transformer decides whether it's a number. Separate NUMBER and word +// terminals overlap, and the lexer takes the first that matches rather +// than the longest, so e.g. an ISO datetime `1997-07-16T19:20:30` lexed as +// `1997` and `-07-16T19:20:30`. +TOKEN: /[a-zA-Z0-9'().\/:;<=>?@_~+\[\\-]+/ COMMA: "," %import common.NEWLINE -> _NL %import common.WS_INLINE %import common.CNAME %import common.WORD -%import common.NUMBER %import common.INT %import common.SH_COMMENT %import common._STRING_INNER diff --git a/flopy4/mf6/codec/reader/transformer/basic.py b/flopy4/mf6/codec/reader/transformer/basic.py index f2468955..6fd596b3 100644 --- a/flopy4/mf6/codec/reader/transformer/basic.py +++ b/flopy4/mf6/codec/reader/transformer/basic.py @@ -1,9 +1,13 @@ +import re from typing import Any from lark import Token, Transformer from flopy4.utils import parse_number +# A whole token that's a number, allowing Fortran `D` exponents (`1D-5`). +_NUMBER = re.compile(r"[+-]?(\d+(\.\d*)?|\.\d+)([eEdD][+-]?\d+)?") + class BasicTransformer(Transformer): """ @@ -41,14 +45,11 @@ def _list(self, items: list[Any]) -> list[Any]: def line(self, items: list[Any]) -> list[Any]: return items - def item(self, items: list[Any]) -> str | float | int: - return items[0] - - def word(self, items: list[Token]) -> str: - return str(items[0]) - - def NUMBER(self, token: Token) -> int | float: - return parse_number(str(token)) + def TOKEN(self, token: Token) -> str | int | float: + value = str(token) + if _NUMBER.fullmatch(value): + return parse_number(value.replace("d", "e").replace("D", "E")) + return value def CNAME(self, token: Token) -> str: return str(token) diff --git a/test/mf6/test_mf6_codec.py b/test/mf6/test_mf6_codec.py index ceae13fc..917125a9 100644 --- a/test/mf6/test_mf6_codec.py +++ b/test/mf6/test_mf6_codec.py @@ -38,6 +38,25 @@ def test_loads_dis_generic_simple(): assert result["griddata"] == [["delr"], ["constant", 100.0], ["delc"], ["constant", 100.0]] +def test_loads_number_only_as_whole_token(): + mf6_input = """ +BEGIN options + start_date_time 1997-07-16T19:20:30.45+01:00 + ts6 filein 1model.ts + x 1,2.5 .5 7. 1e5 3 # comment + y -5 +2 -1.5e3 1D-5 8.2d-4 +END options +""" + + result = loads(mf6_input) + assert result["options"] == [ + ["start_date_time", "1997-07-16T19:20:30.45+01:00"], + ["ts6", "filein", "1model.ts"], + ["x", 1, 2.5, 0.5, 7.0, 1e5, 3], + ["y", -5, 2, -1500.0, 1e-5, 8.2e-4], + ] + + def test_dumps_ic(): from flopy4.mf6.gwf import Dis, Gwf, Ic diff --git a/test/mf6/test_mf6_io_roundtrip.py b/test/mf6/test_mf6_io_roundtrip.py index cbefe711..630dd522 100644 --- a/test/mf6/test_mf6_io_roundtrip.py +++ b/test/mf6/test_mf6_io_roundtrip.py @@ -21,31 +21,6 @@ from .test_mf6_load_all_models import KNOWN_PASSING XFAIL = { - # the basic grammar splits an ISO datetime (TDIS `START_DATE_TIME`) into - # a number and a word, so TDIS fails to write - "tokenizer splits ISO datetimes": { - "mf6/test/test001a_Tharmonic", - "mf6/test/test001a_Tharmonic_tabs", - "mf6/test/test001h_drn_list4", - "mf6/test/test001h_evt_array1", - "mf6/test/test001h_evt_array2", - "mf6/test/test001h_evt_array3", - "mf6/test/test001h_evt_array4", - "mf6/test/test001h_evt_list1", - "mf6/test/test001h_evt_list2", - "mf6/test/test001h_evt_list3", - "mf6/test/test001h_evt_list4", - "mf6/test/test001h_rch_array1", - "mf6/test/test001h_rch_array2", - "mf6/test/test001h_rch_array3", - "mf6/test/test001h_rch_array4", - "mf6/test/test001h_rch_list1", - "mf6/test/test001h_rch_list2", - "mf6/test/test001h_rch_list3", - "mf6/test/test001h_rch_list4", - "mf6/test/test001i_gwf-gwf", - "mf6/test/test001i_multilayer", - }, # array values are written with 9 significant digits "writer rounds floats": { "mf6/test/test033_wtdecay", @@ -86,6 +61,22 @@ # `AUXILIARY ` loads as a str, not a list, and egress drops it "ingress loads a single AUXILIARY name as a str": { "mf6/test/test001a_Tharmonic_extlist", + "mf6/test/test001h_evt_array1", + "mf6/test/test001h_evt_array2", + "mf6/test/test001h_evt_array3", + "mf6/test/test001h_evt_array4", + "mf6/test/test001h_evt_list1", + "mf6/test/test001h_evt_list2", + "mf6/test/test001h_evt_list3", + "mf6/test/test001h_evt_list4", + "mf6/test/test001h_rch_array1", + "mf6/test/test001h_rch_array2", + "mf6/test/test001h_rch_array3", + "mf6/test/test001h_rch_array4", + "mf6/test/test001h_rch_list1", + "mf6/test/test001h_rch_list2", + "mf6/test/test001h_rch_list3", + "mf6/test/test001h_rch_list4", "mf6/test/test005_advgw_tidal", "mf6/test/test201_gwtbuy-henryCHD", "mf6/test/test202_gwtbuy-henryCHDm", @@ -96,6 +87,7 @@ # a TIMEARRAYSERIES-sourced period array is dropped on load (see # test_tas_period_array_kept) "ingress drops TAS period arrays": { + "mf6/test/test001h_rch_array3", "mf6/test/test027_TimeseriesTest", "mf6/test/test027_TimeseriesTest_idomain", }, From ad4c0f4bc9c714c762f11d5042b52cdee8cd677f Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Fri, 25 Sep 2026 09:30:44 -0400 Subject: [PATCH 2/2] lex numbers only as whole tokens in the typed grammar The typed grammar had the same problem as the basic one: a token that started like a number lexed as a number followed by more tokens. TDIS `START_DATE_TIME 1997-07-16T19:20:30.45+01:00` split into `1997`, `-07`, `-16` and `T19:20:30.45+01:00`, and the transformer kept only `1997`. The typed rules branch on whether a token is a number, so it can't lex every token as one terminal like the basic grammar does. Instead INT, SIGNED_INT, NUMBER and SIGNED_NUMBER now have to be followed by whitespace, a comma, a comment or the end of input. Co-Authored-By: Claude Opus 5.5 --- flopy4/mf6/codec/reader/grammar/typed.lark | 13 +++++++++---- test/mf6/test_mf6_reader.py | 20 ++++++++++++++++++++ 2 files changed, 29 insertions(+), 4 deletions(-) diff --git a/flopy4/mf6/codec/reader/grammar/typed.lark b/flopy4/mf6/codec/reader/grammar/typed.lark index 5bc2cdab..421a0cb1 100644 --- a/flopy4/mf6/codec/reader/grammar/typed.lark +++ b/flopy4/mf6/codec/reader/grammar/typed.lark @@ -58,14 +58,19 @@ open_close_redirect: "open/close"i filename [_remark] _NL %import common.WS_INLINE %import common.CNAME %import common.WORD -%import common.NUMBER -%import common.INT %import common.SH_COMMENT %import common.ESCAPED_STRING -%import common.SIGNED_NUMBER -%import common.SIGNED_INT COMMA: "," %ignore WS_INLINE %ignore SH_COMMENT %ignore COMMA + +// common's number terminals, but only as a whole token. Otherwise a token +// that starts like a number lexes as a number followed by a word, e.g. an +// ISO datetime `1997-07-16T19:20:30` as `1997`, `-07`, `-16` and +// `T19:20:30`. +INT: /\d+(?![^\s,#])/ +SIGNED_INT: /[+-]\d+(?![^\s,#])/ +NUMBER: /(\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?(?![^\s,#])/ +SIGNED_NUMBER: /[+-](\d+(\.\d*)?|\.\d+)([eE][+-]?\d+)?(?![^\s,#])/ diff --git a/test/mf6/test_mf6_reader.py b/test/mf6/test_mf6_reader.py index d44e7e34..163d0613 100644 --- a/test/mf6/test_mf6_reader.py +++ b/test/mf6/test_mf6_reader.py @@ -782,3 +782,23 @@ def test_typed_grammar_repeating_block_double_header(tmp_path): result = transformer.transform(parser.parse("BEGIN TIME 1.5\n X CONSTANT 1.0\nEND TIME\n")) assert 1.5 in result["time"] + + +def test_typed_loads_filename_with_leading_digit(dfn_path): + from flopy4.mf6.codec.reader import loads_typed + + result = loads_typed( + "BEGIN options\n ts6 filein 1model.ts\nEND options\n", "gwf-chd", dfn_path=dfn_path + ) + assert result["options"]["ts_filerecord"]["ts6_filename"] == "1model.ts" + + +def test_typed_loads_iso_datetime(dfn_path): + from flopy4.mf6.codec.reader import loads_typed + + result = loads_typed( + "BEGIN options\n start_date_time 1997-07-16T19:20:30.45+01:00\nEND options\n", + "sim-tdis", + dfn_path=dfn_path, + ) + assert result["options"]["start_date_time"] == "1997-07-16T19:20:30.45+01:00"