diff --git a/.github/workflows/documentation.yaml b/.github/workflows/documentation.yaml index 4b3deebb..db11de02 100644 --- a/.github/workflows/documentation.yaml +++ b/.github/workflows/documentation.yaml @@ -3,13 +3,20 @@ name: Documentation check on: [pull_request] -jobs: +permissions: + contents: write + +jobs: docs-checks: name: ${{ matrix.doc-type }} strategy: + # Let every format report. Without this, one failing format cancels + # the other two, and a build that fails partway through the examples + # takes the evidence of the others with it. + fail-fast: false matrix: doc-type: [html, latex, epub] - + runs-on: ubuntu-latest steps: - name: Checkout @@ -28,9 +35,62 @@ jobs: run: | if [ -f requirements.txt ]; then pip install -r requirements.txt; fi pip install -e . + - name: Check the example digest is current + run: python docs/example_digest.py --check + - name: Build ${{ matrix.doc-type }} documentation run: sphinx-build -Wnb ${{ matrix.doc-type }} docs/source/ docs/build-${{ matrix.doc-type }}/ - + + # Read the Docs stops a build at fifteen minutes and the examples take + # about twenty, so it restores what the examples produced here instead + # of running them. The archive is named by the digest of the example + # inputs and of the library, and Read the Docs only restores one whose + # digest matches the commit it is building. + - name: Publish the example cache + if: matrix.doc-type == 'html' && github.event.pull_request.head.repo.full_name == github.repository + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + name=$(python docs/example_cache.py name) + python docs/example_cache.py pack "$name" + gh release view docs-cache >/dev/null 2>&1 \ + || gh release create docs-cache --title "Documentation example cache" \ + --notes "Figures the documentation examples produce, named by the digest from docs/example_digest.py. Read the Docs restores the archive matching the commit it builds. Managed by the documentation workflow." \ + --latest=false + gh release upload docs-cache "$name" --clobber + + # A pull request that changes an example has no cache for its new + # digest until the step above finishes, so the Read the Docs build + # that started alongside this job ran the examples and timed out. + # Build it again now that the archive is published, and it is warm. + # + # Needs a READTHEDOCS_TOKEN secret, from Settings -> API Tokens on + # Read the Docs. Without it this step says so and does nothing, and + # the only cost is having to press Rebuild by hand. + - name: Rebuild the docs on Read the Docs + if: matrix.doc-type == 'html' && github.event.pull_request.head.repo.full_name == github.repository + env: + RTD_TOKEN: ${{ secrets.READTHEDOCS_TOKEN }} + RTD_VERSION: ${{ github.event.pull_request.number }} + run: | + if [ -z "$RTD_TOKEN" ]; then + echo "No READTHEDOCS_TOKEN secret, skipping the rebuild." + echo "Press Rebuild on the Read the Docs build to pick up the cache." + exit 0 + fi + # A pull request build is a version named after the pull request. + code=$(curl -sS -o /tmp/rtd.json -w '%{http_code}' -X POST \ + -H "Authorization: Token $RTD_TOKEN" \ + "https://app.readthedocs.org/api/v3/projects/microstructpy/versions/$RTD_VERSION/builds/") + echo "Read the Docs answered $code" + cat /tmp/rtd.json + case "$code" in + 202) echo "Rebuild triggered." ;; + *) echo "Could not trigger a rebuild. The cache is published, so a" + echo "rebuild by hand will be warm."; exit 0 ;; + esac + + - name: Prepare documentation artifact run: | # Define Path to Upload diff --git a/.readthedocs.yaml b/.readthedocs.yaml index d71c95c1..bc732148 100644 --- a/.readthedocs.yaml +++ b/.readthedocs.yaml @@ -13,6 +13,30 @@ build: python: "3.10" apt_packages: - freeglut3-dev + jobs: + # Running every example takes about twenty minutes, and a build here + # is stopped at fifteen. The documentation workflow runs them on a + # machine without that limit and publishes what they produced, keyed + # by a digest of the example inputs and of the library. Restore that + # and Sphinx-Gallery skips the examples. + # + # The archive is only restored when its digest is the digest of this + # commit, so an edited example is never served with older figures. + # When there is no archive to match, nothing is restored, the + # examples run, and the build is likely to time out. That is the + # intended failure: publish the cache for this commit, then rebuild. + post_install: + - | + set -u + name=$(python docs/example_cache.py name) || exit 0 + url=https://github.com/kip-hart/MicroStructPy/releases/download/docs-cache + echo "Looking for $name" + if curl -sSfL -o /tmp/examples.tar.gz "$url/$name"; then + python docs/example_cache.py restore /tmp/examples.tar.gz \ + || echo "The archive was refused, the examples will be run." + else + echo "No archive published for this commit, the examples will be run." + fi # Build documentation in the "docs/" directory with Sphinx sphinx: diff --git a/docs/README.rst b/docs/README.rst new file mode 100644 index 00000000..989d0445 --- /dev/null +++ b/docs/README.rst @@ -0,0 +1,67 @@ +Building the documentation +========================== + +:: + + pip install -r docs/requirements.txt + pip install -r requirements.txt + pip install -e . + sphinx-build -Wnb html docs/source docs/build-html + +The first build takes about twenty minutes, because +``docs/source/sphinx_gallery/plot_demos.py`` runs every example in +``src/microstructpy/examples`` to produce the figures the pages embed. +Later builds reuse that work and take seconds. + + +The example cache +----------------- + +A build on Read the Docs is stopped at fifteen minutes on the free plan, +which is less than the examples take. The documentation workflow runs +them on a machine without that limit and publishes what they produced, +and the Read the Docs build restores it instead of running them. + +Two pieces make that safe. + +``docs/example_digest.py`` digests the example inputs and the library +source, and keeps the result in ``plot_demos.py`` as ``EXAMPLES_DIGEST``. +Sphinx-Gallery decides whether to run a script by hashing it, and +``plot_demos.py`` finds the examples with ``glob`` rather than naming +them, so without this its hash would not move when an example was edited +or added, and a cached build would serve stale figures. Editing an +example now changes the digest, which changes the hash, which runs the +examples again. + +``docs/example_cache.py`` packs and restores the figures. An archive +carries the digest it was built from, and ``restore`` refuses one that +does not match the working tree. A build that cannot be warm runs the +examples and may time out, which is intended: it is better to be slow and +correct than fast and stale. + +After editing an example or anything in ``src/microstructpy``:: + + python docs/example_digest.py --write + +The documentation check runs ``python docs/example_digest.py --check`` and +fails if the stored digest is out of date. + + +Publishing a cache by hand +-------------------------- + +The workflow does this on every pull request, from the ``html`` job. To do +it manually, after a full build:: + + name=$(python docs/example_cache.py name) + python docs/example_cache.py pack "$name" + gh release upload docs-cache "$name" --clobber + +Read the Docs fetches the archive for the digest of the commit it is +building, so a commit whose examples and library are unchanged reuses the +archive of an earlier one. + +A pull request that changes an example has no cache for its new digest +until its ``html`` job finishes. A Read the Docs build that starts before +that will run the examples and time out. Rebuild it once the job has +published the archive. diff --git a/docs/example_cache.py b/docs/example_cache.py new file mode 100644 index 00000000..0f94a317 --- /dev/null +++ b/docs/example_cache.py @@ -0,0 +1,193 @@ +#!/usr/bin/env python3 +"""Pack and restore the figures the documentation examples produce. + +The documentation build runs every example in +``src/microstructpy/examples``, which takes about twenty minutes and does +not fit in the fifteen minute limit of a Read the Docs build on the free +plan. The work is the same on every build, so this module moves it to a +machine without a time limit: the documentation workflow packs what the +examples produced, and the Read the Docs build restores it before Sphinx +runs. + +Two sets of files are needed, and both are in the archive: + +* ``docs/source/auto_examples``, the output of Sphinx-Gallery. It holds + the ``.md5`` of ``plot_demos.py``, which is what makes Sphinx-Gallery + skip the examples instead of running them. +* The PNG files under ``src/microstructpy/examples``, which the pages in + ``docs/source/examples`` embed with ``figure::``. The meshes beside + them are not in the archive, since no page refers to them. + +An archive carries the digest of the inputs it was built from, from +:mod:`docs.example_digest`. ``restore`` refuses an archive whose digest +is not the digest of the working tree, so an edited example cannot be +served with the figures of an older one. Refusing leaves the build to +run the examples itself, which is slow and may time out, and that is the +intended outcome: a build that cannot be warm should be loud, not stale. + +Usage:: + + python docs/example_cache.py pack examples.tar.gz + python docs/example_cache.py restore examples.tar.gz + python docs/example_cache.py name # archive name for this tree + +""" + +import argparse +import json +import os +import sys +import tarfile + +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.dirname(HERE) + +sys.path.insert(0, HERE) +from example_digest import digest # noqa: E402 + +GALLERY = os.path.join('docs', 'source', 'auto_examples') +EXAMPLES = os.path.join('src', 'microstructpy', 'examples') +MANIFEST = 'microstructpy-example-cache.json' + + +def _gallery_members(): + """Every file of the Sphinx-Gallery output, relative to the root.""" + members = [] + root = os.path.join(ROOT, GALLERY) + for dirpath, dirnames, filenames in os.walk(root): + dirnames.sort() + for name in sorted(filenames): + path = os.path.join(dirpath, name) + members.append(os.path.relpath(path, ROOT)) + return members + + +def _figure_members(): + """Every PNG an example wrote, relative to the root. + + Only the files in the output directories of the examples, so that an + input such as ``aluminum_micro.png`` is not carried in the archive. + """ + members = [] + root = os.path.join(ROOT, EXAMPLES) + for dirpath, dirnames, filenames in os.walk(root): + dirnames.sort() + if os.path.abspath(dirpath) == os.path.abspath(root): + continue # the inputs live here, the figures are in subdirs + for name in sorted(filenames): + if name.endswith('.png'): + path = os.path.join(dirpath, name) + members.append(os.path.relpath(path, ROOT)) + return members + + +def pack(filename): + """Write an archive of the figures and the gallery output.""" + gallery = _gallery_members() + figures = _figure_members() + + if not gallery: + print('No %s to pack. Build the documentation first.' % GALLERY, + file=sys.stderr) + return 1 + if not figures: + print('No figures under %s to pack.' % EXAMPLES, file=sys.stderr) + return 1 + + value = digest() + manifest = json.dumps({'digest': value, + 'gallery': len(gallery), + 'figures': len(figures)}, indent=2).encode() + + with tarfile.open(filename, 'w:gz') as tar: + info = tarfile.TarInfo(MANIFEST) + info.size = len(manifest) + tar.addfile(info, _BytesIO(manifest)) + for rel in gallery + figures: + tar.add(os.path.join(ROOT, rel), arcname=rel) + + size = os.path.getsize(filename) / 1048576.0 + print('packed %d gallery files and %d figures for digest %s ' + 'into %s (%.1f MB)' + % (len(gallery), len(figures), value, filename, size)) + return 0 + + +def restore(filename): + """Extract an archive, if its digest is the digest of this tree.""" + if not os.path.exists(filename): + print('No archive at %s.' % filename, file=sys.stderr) + return 1 + + with tarfile.open(filename, 'r:gz') as tar: + try: + manifest = json.loads(tar.extractfile(MANIFEST).read().decode()) + except KeyError: + print('%s has no %s, refusing to restore it.' + % (filename, MANIFEST), file=sys.stderr) + return 1 + + current = digest() + if manifest.get('digest') != current: + print('The archive was built from %s and the examples digest ' + 'to %s. Refusing to restore it, the examples will be run.' + % (manifest.get('digest'), current), file=sys.stderr) + return 1 + + members = [m for m in tar.getmembers() if m.name != MANIFEST] + for m in members: + if m.name.startswith('/') or '..' in m.name.split('/'): + print('Refusing a path outside the tree: %s' % m.name, + file=sys.stderr) + return 1 + try: + tar.extractall(ROOT, members=members, filter='data') + except TypeError: + # filter= arrived in Python 3.12. The paths are checked above. + tar.extractall(ROOT, members=members) + + print('restored %d files for digest %s' % (len(members), current)) + return 0 + + +def name(): + """Print the archive name for this working tree.""" + print('microstructpy-examples-%s.tar.gz' % digest()) + return 0 + + +class _BytesIO(object): + """A minimal file object, so tarfile can add bytes already in memory.""" + + def __init__(self, data): + self._data = data + self._pos = 0 + + def read(self, size=-1): + if size < 0: + chunk = self._data[self._pos:] + else: + chunk = self._data[self._pos:self._pos + size] + self._pos += len(chunk) + return chunk + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.split('\n')[0]) + sub = parser.add_subparsers(dest='command', required=True) + p = sub.add_parser('pack', help='write an archive of the figures') + p.add_argument('filename') + r = sub.add_parser('restore', help='extract an archive of the figures') + r.add_argument('filename') + sub.add_parser('name', help='the archive name for this working tree') + + args = parser.parse_args() + if args.command == 'pack': + return pack(args.filename) + if args.command == 'restore': + return restore(args.filename) + return name() + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/docs/example_digest.py b/docs/example_digest.py new file mode 100644 index 00000000..c736ea8c --- /dev/null +++ b/docs/example_digest.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Digest of everything that changes the output of the documentation examples. + +The documentation build runs every example through +``docs/source/sphinx_gallery/plot_demos.py``, which takes about twenty +minutes. Sphinx-Gallery skips an example whose source file still matches +the ``.md5`` stored next to its generated output, so a build that starts +from the output of an earlier one does no work at all. + +``plot_demos.py`` finds the examples with :func:`glob.glob`, and names +none of them, so its own hash does not move when an example is edited or +added, nor when the library that meshes them changes. Caching on that +hash alone would serve stale figures and never say so. + +This module closes that gap. It digests the example inputs and the +library source, and keeps the result in ``plot_demos.py`` as the value of +``EXAMPLES_DIGEST``. Editing an example changes the digest, which changes +the hash of ``plot_demos.py``, which makes Sphinx-Gallery run the examples +again. + +Usage:: + + python docs/example_digest.py # print the digest + python docs/example_digest.py --write # update plot_demos.py + python docs/example_digest.py --check # exit 1 if it is out of date + +""" + +import argparse +import hashlib +import os +import re +import sys + +HERE = os.path.dirname(os.path.abspath(__file__)) +ROOT = os.path.dirname(HERE) +EXAMPLE_DIR = os.path.join(ROOT, 'src', 'microstructpy', 'examples') +PACKAGE_DIR = os.path.join(ROOT, 'src', 'microstructpy') +DEMOS = os.path.join(HERE, 'source', 'sphinx_gallery', 'plot_demos.py') + +#: Inputs of the examples. Everything the examples read. +INPUT_SUFFIXES = ('.xml', '.py', '.csv') + +#: The line that holds the digest in plot_demos.py. +DIGEST_RE = re.compile(r"^EXAMPLES_DIGEST = '[0-9a-f]*'$", re.M) + + +def _digest_files(): + """The files the digest covers, as absolute paths, sorted. + + The example inputs, and the library that turns them into figures. The + output directories of the examples are not included: they are what the + digest is about to describe. + """ + paths = [] + + for name in sorted(os.listdir(EXAMPLE_DIR)): + path = os.path.join(EXAMPLE_DIR, name) + if os.path.isfile(path) and name.endswith(INPUT_SUFFIXES): + paths.append(path) + + for dirpath, dirnames, filenames in os.walk(PACKAGE_DIR): + dirnames.sort() + if os.path.abspath(dirpath).startswith(os.path.abspath(EXAMPLE_DIR)): + continue + for name in sorted(filenames): + if name.endswith('.py'): + paths.append(os.path.join(dirpath, name)) + + return sorted(set(paths)) + + +def digest(): + """The digest of the example inputs and the library. + + Returns: + str: The first 16 characters of the SHA-256 of the files, each + entered by its path relative to the root of the repository and by + its contents. + """ + sha = hashlib.sha256() + for path in _digest_files(): + rel = os.path.relpath(path, ROOT).replace(os.sep, '/') + sha.update(rel.encode('utf-8')) + sha.update(b'\0') + with open(path, 'rb') as f: + sha.update(f.read()) + sha.update(b'\0') + return sha.hexdigest()[:16] + + +def stored_digest(): + """The digest currently written in plot_demos.py, or an empty string.""" + with open(DEMOS, 'r') as f: + match = DIGEST_RE.search(f.read()) + return match.group(0).split("'")[1] if match else '' + + +def write_digest(value): + """Write a digest into plot_demos.py. Returns True if it changed.""" + with open(DEMOS, 'r') as f: + text = f.read() + new = DIGEST_RE.sub("EXAMPLES_DIGEST = '%s'" % value, text) + if new == text: + return False + with open(DEMOS, 'w') as f: + f.write(new) + return True + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.split('\n')[0]) + group = parser.add_mutually_exclusive_group() + group.add_argument('--write', action='store_true', + help='update the digest in plot_demos.py') + group.add_argument('--check', action='store_true', + help='exit 1 if the stored digest is out of date') + args = parser.parse_args() + + current = digest() + + if args.write: + if write_digest(current): + print('updated EXAMPLES_DIGEST to %s' % current) + else: + print('EXAMPLES_DIGEST already %s' % current) + return 0 + + if args.check: + stored = stored_digest() + if stored == current: + print('EXAMPLES_DIGEST is up to date (%s)' % current) + return 0 + print('EXAMPLES_DIGEST is %s, the examples digest to %s.' + % (stored or 'missing', current), file=sys.stderr) + print('Run: python docs/example_digest.py --write', file=sys.stderr) + return 1 + + print(current) + return 0 + + +if __name__ == '__main__': + sys.exit(main()) diff --git a/docs/source/sphinx_gallery/plot_demos.py b/docs/source/sphinx_gallery/plot_demos.py index 51d2f197..1e329bae 100644 --- a/docs/source/sphinx_gallery/plot_demos.py +++ b/docs/source/sphinx_gallery/plot_demos.py @@ -15,6 +15,17 @@ locale.setlocale(locale.LC_NUMERIC, "C") +# Digest of the example inputs and of the library that meshes them. +# +# Sphinx-Gallery skips this script when its hash matches the .md5 beside +# the output of an earlier build, which is what lets a cached build skip +# the examples. The examples are found by glob below and none of them are +# named here, so without this line the hash would not move when one of +# them is edited, and the build would serve stale figures. +# +# Managed by docs/example_digest.py. Do not edit by hand. +EXAMPLES_DIGEST = '314fadcaec7cb6e5' + example_dir = '../../../src/microstructpy/examples' welcome_fnames = ['intro_2_quality/trimesh.png', diff --git a/src/microstructpy/examples/basalt_circle.xml b/src/microstructpy/examples/basalt_circle.xml index 6ce1d39e..110fa271 100644 --- a/src/microstructpy/examples/basalt_circle.xml +++ b/src/microstructpy/examples/basalt_circle.xml @@ -142,7 +142,7 @@ True True - 0.01 + 0.02 20 0.05 diff --git a/src/microstructpy/examples/elliptical_grains.xml b/src/microstructpy/examples/elliptical_grains.xml index 7bd02ee6..2219a88a 100644 --- a/src/microstructpy/examples/elliptical_grains.xml +++ b/src/microstructpy/examples/elliptical_grains.xml @@ -1,4 +1,4 @@ - + 2