diff --git a/README.md b/README.md index c8f46c3..fbd9d28 100644 --- a/README.md +++ b/README.md @@ -224,6 +224,7 @@ config = sbd.TPB_SBD() | `adet_comm_size` | 1 | Alpha determinant communicator size | | `bdet_comm_size` | 1 | Beta determinant communicator size | | `task_comm_size` | 1 | Task communicator size | +| `dump_matrix_form_wf` | `""` | Path to write the amplitudes to — see [Getting the TPB amplitudes](#getting-the-tpb-amplitudes) for the file format | Total MPI ranks = `task_comm_size × adet_comm_size × bdet_comm_size`. @@ -267,6 +268,39 @@ results = sbd.tpb_diag(fcidump, adet, bdet, sbd_data, **Returns:** `dict` with keys `energy`, `density`, `carryover_adet`, `carryover_bdet`, `one_p_rdm`, `two_p_rdm`. +#### Getting the TPB amplitudes + +Neither return dict holds the wavefunction amplitudes. Two separate mechanisms write them, in **different formats**, and the one named in the function signature is not the one most callers want: + +| | `sbd_data.dump_matrix_form_wf` | `savename` argument | +|---|---|---| +| Writes | the full `len(adet) × len(bdet)` subspace, gathered onto one rank | one file per b_comm position, each holding **only that rank's block** | +| Path | exactly what you set | `f"{savename}{rank_b:06d}.bin"` | +| Header | **none** | three `size_t` (`adet_range`, `bdet_range`, `det_length`), then that block's alpha and beta determinant words | +| Payload | `float64`, row-major over α×β | `adet_range × bdet_range` `float64` | +| Intended for | analysis | checkpoint/restart, paired with `loadname` | + +So the whole-subspace read is just: + +```python +config.dump_matrix_form_wf = "wf.bin" +results = sbd.tpb_diag(fcidump, adet, bdet, config) + +amplitudes = np.fromfile("wf.bin").reshape(len(adet), len(bdet)) +``` + +Naming that file `.dat` or `.txt` instead writes a text table labelling every amplitude with its own alpha and beta bitstring — the easiest form to inspect by hand, and it sidesteps having to reconstruct the ordering yourself: + +``` + # : : +``` + +Be aware that this text form is written with the stream's default formatting, so amplitudes land with only ~6 significant digits; use the binary form when you need full `float64` precision. + +Rows and columns follow `adet`/`bdet` as the diagonalizer saw them. `sort_bitarray` removes duplicates as well as sorting, so pass lists already in that form if you intend to label the axes yourself — `from_strings` documents the word packing. + +This is the mechanism [`sbd_solver.py`](python/sbd_solver.py) uses to populate `SCIState.amplitudes` — though it skips the write entirely when SBD is picking the carryover itself (`carryover_type` 1, against a qiskit-addon-sqd that accepts it), since then the next subspace comes back in `carryover_adet`/`carryover_bdet` and the amplitudes never need to reach Python. + ```python # GDB: over an explicit list of full determinants rather than a product space results = sbd.gdb_diag(fcidump, det, sbd_data, diff --git a/python/__init__.py b/python/__init__.py index f8ccab8..fce06f9 100644 --- a/python/__init__.py +++ b/python/__init__.py @@ -526,13 +526,23 @@ def LoadAlphaDets(filename, bit_length, total_bit_length, device=None): def makestring(config, bit_length, total_bit_length, device=None): - """Convert determinant to string representation.""" + """Render a packed determinant as a ``total_bit_length``-character string. + + The inverse of :func:`from_string`, and subject to the same right-to-left + word packing, which :func:`from_strings` describes. + """ _ensure_initialized() return get_backend(device).makestring(config, bit_length, total_bit_length) def from_string(s, bit_length, total_bit_length, device=None): - """Convert binary string to determinant format.""" + """Convert a binary string to SBD's packed determinant format. + + Returns a list of ``size_t`` words, e.g. ``from_string("0011", 20, 4)`` is + ``[3]``. :func:`makestring` is the inverse. See :func:`from_strings`, the + bulk form, for the word-packing convention and for why it is worth + preferring that form outside of one-off calls. + """ _ensure_initialized() return get_backend(device).from_string(s, bit_length, total_bit_length) @@ -544,6 +554,25 @@ def from_strings(strings, bit_length, total_bit_length, device=None): determinant across the Python boundary. Prefer it for anything larger than a handful: the per-call form costs tens of thousands of round trips on the vendored inputs alone. + + Each row holds ``total_bit_length`` bits packed ``bit_length`` to a word, so + a bitstring longer than ``bit_length`` occupies more than one word. The + packing runs from the **right end** of the string: its trailing + ``bit_length`` characters become word 0, the next ``bit_length`` to the left + become word 1, and so on, so the leading characters of a multi-word string + land in the *last* word rather than the first:: + + from_strings(["0011"], 20, 4) # -> array([[3]], dtype=uint64) + from_strings(["1" + "0" * 69], 63, 70) # -> array([[ 0, 64]], ...) + from_strings(["0" * 69 + "1"], 63, 70) # -> array([[1, 0]], ...) + + Within a word the rightmost character is the least significant bit, which is + why ``"0011"`` packs to ``3`` rather than ``12``. Getting this backwards + produces a valid-looking but wrong subspace rather than an error, so check + round trips with :func:`makestring` when building determinants by hand. + + ``bit_length`` must match the ``bit_length`` of the config passed to the + diagonalizers, since it sets how many words each determinant occupies. """ _ensure_initialized() return get_backend(device).from_strings(list(strings), bit_length, total_bit_length) @@ -579,13 +608,17 @@ def tpb_diag_from_files(fcidumpfile, adetfile, sbd_data, fcidumpfile: Path to FCIDUMP file. adetfile: Path to alpha determinants file. sbd_data: TPB_SBD configuration object. - loadname: Path to load initial wavefunction (optional). - savename: Path to save final wavefunction (optional). + loadname: Path prefix to load an initial wavefunction from (optional). + savename: Path prefix to save the final wavefunction to (optional). device: Override device ('cpu', 'gpu', or None for default). Returns: dict with keys: energy, density, carryover_adet, carryover_bdet, one_p_rdm, two_p_rdm. + + Note: + As with :func:`tpb_diag`, no amplitudes are returned in the dict; see + that function for the two ways of writing them to disk. """ _ensure_initialized() backend = get_backend(device) @@ -604,13 +637,42 @@ def tpb_diag(fcidump, adet, bdet, sbd_data, adet: Alpha determinants. bdet: Beta determinants. sbd_data: TPB_SBD configuration object. - loadname: Path to load initial wavefunction (optional). - savename: Path to save final wavefunction (optional). + loadname: Path prefix to load an initial wavefunction from (optional). + savename: Path prefix to save the final wavefunction to (optional). See + the note below: for reading amplitudes back, + ``sbd_data.dump_matrix_form_wf`` is usually the one you want. device: Override device ('cpu', 'gpu', or None for default). Returns: dict with keys: energy, density, carryover_adet, carryover_bdet, one_p_rdm, two_p_rdm. + + Note: + The returned dict holds no wavefunction amplitudes. There are two ways + to get them, and they write *different* formats -- do not mix them up: + + ``sbd_data.dump_matrix_form_wf = "wf.bin"`` writes the full + ``len(adet) x len(bdet)`` subspace, gathered onto one rank, as a bare + row-major array of ``float64`` with **no header** -- so + ``np.fromfile(path).reshape(len(adet), len(bdet))`` reads it. Naming it + ``.dat`` or ``.txt`` instead writes a text table in which every + amplitude is labelled with its own alpha and beta bitstring, which is + the easiest form to inspect by hand. This is the mechanism this + package's own SQD integration uses. + + ``savename`` instead writes one file per b_comm position, + ``f"{savename}{rank_b:06d}.bin"``, each holding only **that rank's + block** of the subspace: three ``size_t`` headers + (``adet_range``, ``bdet_range``, ``det_length``), then that block's + alpha and then beta determinant words, then + ``adet_range x bdet_range`` ``float64`` amplitudes. It pairs with + ``loadname`` for restarting a run, so it is the better choice for + checkpointing and the worse one for analysis. + + Amplitudes are ordered by ``adet``/``bdet`` as the diagonalizer saw + them. ``sort_bitarray`` removes duplicates as well as sorting, so pass + lists already in that form if you need to label the rows and columns + yourself. """ _ensure_initialized() backend = get_backend(device) diff --git a/python/bindings.cpp b/python/bindings.cpp index 19fbb4b..266d3a6 100644 --- a/python/bindings.cpp +++ b/python/bindings.cpp @@ -319,7 +319,12 @@ PYBIND11_MODULE(SBD_MODULE_NAME, m) { .def_readwrite("bit_length", &sbd::tpb::SBD::bit_length, "Bit length for determinant representation") .def_readwrite("dump_matrix_form_wf", &sbd::tpb::SBD::dump_matrix_form_wf, - "Filename to dump wavefunction in matrix form") + "Path to write the full adet x bdet amplitude matrix to, " + "gathered onto one rank. A .bin path writes a bare " + "row-major float64 array with no header; .dat/.txt write " + "a labelled text table. Distinct from tpb_diag's " + "savename, which writes per-rank blocks with a header " + "for restart -- see the tpb_diag docstring.") #ifdef SBD_THRUST .def_readwrite("use_precalculated_dets", &sbd::tpb::SBD::use_precalculated_dets, "Use precalculated determinants (THRUST)") diff --git a/releasenotes/notes/document-tpb-amplitude-output-2d9f4a1c76e8b305.yaml b/releasenotes/notes/document-tpb-amplitude-output-2d9f4a1c76e8b305.yaml new file mode 100644 index 0000000..c45b61e --- /dev/null +++ b/releasenotes/notes/document-tpb-amplitude-output-2d9f4a1c76e8b305.yaml @@ -0,0 +1,20 @@ +--- +other: + - | + TPB's two wavefunction output paths are now documented, including the fact + that they write different formats. ``sbd_data.dump_matrix_form_wf`` writes + the full ``len(adet) x len(bdet)`` subspace, gathered onto one rank, as a + bare row-major ``float64`` array with no header, so + ``np.fromfile(path).reshape(len(adet), len(bdet))`` reads it; naming the + file ``.dat`` or ``.txt`` writes a text table labelling each amplitude with + its own alpha and beta bitstring, at the stream's default precision rather + than full ``float64``. The ``savename`` argument of ``tpb_diag`` instead + writes one file per b_comm position holding only that rank's block, behind a + three-``size_t`` header and that block's determinant words, and is meant to + pair with ``loadname`` for restart. For reading amplitudes back, + ``dump_matrix_form_wf`` is normally the one you want. + + The packing convention of ``from_string`` and ``makestring`` is documented + as well. Bitstrings are packed from the right: the trailing ``bit_length`` + characters become word 0, so the leading characters of a string longer than + one word land in the last word rather than the first.