From 719aec7fe7bbbe4b527d82f2f31498df21e43d94 Mon Sep 17 00:00:00 2001 From: Tom Vo Date: Mon, 5 Oct 2026 13:21:48 -0500 Subject: [PATCH 1/2] [DevOps]: Schedule diagnostics linking on Chrysalis and Spin --- .../ingestion/diagnostics_link_scanner.py | 179 ++++++++++++---- .../sites/site_ingestion_launcher.sh | 110 ++++++---- .../ingestion/sites/templates/crontab.example | 13 ++ .../test_diagnostics_link_scanner.py | 194 +++++++++++++++++- .../test_site_collection_launcher.py | 126 +++++++++++- docs/operations/nersc-spin-runbook.md | 111 +++++++++- docs/operations/setup-ingestion-operations.md | 63 ++++-- docs/operations/test-ingestion-operations.md | 16 ++ 8 files changed, 698 insertions(+), 114 deletions(-) diff --git a/backend/app/scripts/ingestion/diagnostics_link_scanner.py b/backend/app/scripts/ingestion/diagnostics_link_scanner.py index 79ae5032..7cb43b6d 100644 --- a/backend/app/scripts/ingestion/diagnostics_link_scanner.py +++ b/backend/app/scripts/ingestion/diagnostics_link_scanner.py @@ -7,6 +7,7 @@ import posixpath import re import time +import traceback from dataclasses import dataclass from datetime import datetime, timezone from pathlib import Path @@ -36,17 +37,27 @@ class Candidate: fingerprint: str -def run(included_case_paths: set[Path] | None = None) -> int: - """Scan the archive, optionally limited to root-relative case paths.""" - configured_machine = os.environ.get("MACHINE_NAME", "").strip() - if not configured_machine: - raise ValueError("MACHINE_NAME is required") +@dataclass +class ScanProgress: + """Track the active phase and candidate for fatal-error reporting.""" - machine = canonicalize_machine_name(configured_machine) - archive = _resolve_archive(machine) - root = Path(archive.root) - dry_run = os.environ.get("DRY_RUN", "true").lower() in {"1", "true", "yes"} + phase: str = "configuration" + candidate_path: str | None = None + +@dataclass(frozen=True) +class ScannerConfig: + """Runtime settings shared by discovery and linking.""" + + machine: str + archive: DiagnosticsArchive + dry_run: bool + api_base: str + token: str + + +def run(included_case_paths: set[Path] | None = None) -> int: + """Scan optional root-relative case paths and always report exit counters.""" summary = { "discovered_candidates": 0, "dry_run_candidates": 0, @@ -55,11 +66,50 @@ def run(included_case_paths: set[Path] | None = None) -> int: "submitted_links": 0, "failed_link_submissions": 0, } + progress = ScanProgress() + outcome = "failed" + try: + result = _run_scan(summary, progress, included_case_paths) + outcome = "completed" if result == 0 else "failed" + return result + except Exception as exc: + # Do not log exception messages/locals: HTTP errors may contain URLs, + # credentials or response bodies. Keep frame locations for diagnosis. + frames = traceback.extract_tb(exc.__traceback__) + _log_event( + "diagnostics_scanner_failed", + { + "phase": progress.phase, + "archive_relative_case_path": progress.candidate_path, + "error_type": type(exc).__name__, + "traceback_frames": [ + f"{Path(frame.filename).name}:{frame.lineno}:{frame.name}" + for frame in frames + ], + }, + ) + return 1 + finally: + _log_event("diagnostics_scanner_completed", {**summary, "outcome": outcome}) + + +def _build_scanner_config() -> ScannerConfig: + """Resolve the reviewed archive and validate the scheduler settings.""" + configured_machine = os.environ.get("MACHINE_NAME", "").strip() + if not configured_machine: + raise ValueError("MACHINE_NAME is required") + + machine = canonicalize_machine_name(configured_machine) + archive = _resolve_archive(machine) + dry_run_value = os.environ.get("DRY_RUN", "true").strip().lower() + if dry_run_value not in {"1", "true", "yes", "on", "0", "false", "no", "off"}: + raise ValueError("DRY_RUN must be a boolean") + dry_run = dry_run_value in {"1", "true", "yes", "on"} _log_event( "diagnostics_scanner_startup_configuration", { "machine_name": machine, - "archive_root": str(root), + "archive_root": archive.root, "public_base_url": _sanitize_url(archive.public_base_url), "dry_run": dry_run, "has_api_base_url": bool(os.environ.get("SIMBOARD_API_BASE_URL")), @@ -67,39 +117,91 @@ def run(included_case_paths: set[Path] | None = None) -> int: }, ) - candidates = _discover(root, archive.public_base_url, machine, included_case_paths) - summary["discovered_candidates"] = len(candidates) - _log_event("diagnostics_scanner_discovery_completed", summary.copy()) + api_base = "" + token = "" + if not dry_run: + api_base = os.environ["SIMBOARD_API_BASE_URL"].rstrip("/") + token = os.environ["SIMBOARD_API_TOKEN"] + if not api_base or not token: + raise ValueError("Live scans require API configuration") + return ScannerConfig( + machine=machine, + archive=archive, + dry_run=dry_run, + api_base=api_base, + token=token, + ) - if dry_run: - for candidate in candidates: - relative = candidate.path.parent.relative_to(root).as_posix() - summary["dry_run_candidates"] += 1 - _log_event( - "diagnostics_scanner_dry_run_candidate", - { - "archive_relative_case_path": relative, - "settings_filename": candidate.settings.name, - "fingerprint": candidate.fingerprint, - }, - ) - _log_event("diagnostics_scanner_dry_run_completed", summary.copy()) - _log_event("diagnostics_scanner_completed", summary.copy()) - return 0 +def _run_scan( + summary: dict[str, int], + progress: ScanProgress, + included_case_paths: set[Path] | None = None, +) -> int: + """Discover candidates, then log a dry run or link them through the API.""" + config = _build_scanner_config() + root = Path(config.archive.root) + progress.phase = "discovery" + candidates = _discover( + root, config.archive.public_base_url, config.machine, included_case_paths + ) + summary["discovered_candidates"] = len(candidates) + _log_event("diagnostics_scanner_discovery_completed", summary.copy()) + if config.dry_run: + _log_dry_run(candidates, root, summary, progress) + else: + _link_candidates(candidates, config, summary, progress) + if summary["deferred_state_lookups"] or summary["failed_link_submissions"]: + return 1 + return 0 - api_base = os.environ["SIMBOARD_API_BASE_URL"].rstrip("/") - token = os.environ["SIMBOARD_API_TOKEN"] - headers = {"Authorization": f"Bearer {token}"} +def _log_dry_run( + candidates: list[Candidate], + root: Path, + summary: dict[str, int], + progress: ScanProgress, +) -> None: + """Report discovered candidates without opening an API client.""" + progress.phase = "dry_run" + for candidate in candidates: + relative = candidate.path.parent.relative_to(root).as_posix() + progress.candidate_path = relative + summary["dry_run_candidates"] += 1 + _log_event( + "diagnostics_scanner_dry_run_candidate", + { + "archive_relative_case_path": relative, + "settings_filename": candidate.settings.name, + "fingerprint": candidate.fingerprint, + }, + ) + _log_event("diagnostics_scanner_dry_run_completed", summary.copy()) + + +def _link_candidates( + candidates: list[Candidate], + config: ScannerConfig, + summary: dict[str, int], + progress: ScanProgress, +) -> None: + """Skip unchanged candidates and submit new links, counting API failures.""" + root = Path(config.archive.root) + headers = {"Authorization": f"Bearer {config.token}"} + progress.phase = "api_client" with httpx.Client(timeout=30) as client: for candidate in candidates: relative = candidate.path.parent.relative_to(root).as_posix() + progress.candidate_path = relative + progress.phase = "state_lookup" state = _request_with_retry( client.get, - f"{api_base}/api/v1/diagnostics/scanner-state", - params={"machine": machine, "archive_relative_case_path": relative}, + f"{config.api_base}/api/v1/diagnostics/scanner-state", + params={ + "machine": config.machine, + "archive_relative_case_path": relative, + }, headers=headers, ) _log_event( @@ -137,6 +239,7 @@ def run(included_case_paths: set[Path] | None = None) -> int: ) continue + progress.phase = "link_submission" payload = { "caseName": candidate.values["case_name"], "machine": candidate.values["machine"], @@ -158,11 +261,10 @@ def run(included_case_paths: set[Path] | None = None) -> int: response = _request_with_retry( client.post, - f"{api_base}/api/v1/diagnostics/scanner/link", + f"{config.api_base}/api/v1/diagnostics/scanner/link", json=payload, headers=headers, ) - if response is None or response.status_code != 204: summary["failed_link_submissions"] += 1 _log_event( @@ -184,13 +286,6 @@ def run(included_case_paths: set[Path] | None = None) -> int: }, ) - _log_event("diagnostics_scanner_completed", summary) - - if summary["deferred_state_lookups"] or summary["failed_link_submissions"]: - return 1 - - return 0 - def _resolve_archive(machine_name: str) -> DiagnosticsArchive: canonical_machine_name = canonicalize_machine_name(machine_name) diff --git a/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh b/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh index ec23a971..a5c12fd7 100755 --- a/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh +++ b/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh @@ -5,9 +5,9 @@ umask 027 # ============================================================================= # Command-line input # ============================================================================= -# Accept one configured site and one supported archive scan mode. -if (( $# != 2 )) || [[ ! $1 =~ ^[a-z0-9_-]+$ ]] || [[ $2 != "archive" && $2 != "staging" ]]; then - echo "Usage: $0 " >&2 +# Accept one configured site and one supported operation. +if (( $# != 2 )) || [[ ! $1 =~ ^[a-z0-9_-]+$ ]] || [[ $2 != "archive" && $2 != "staging" && $2 != "diagnostics" ]]; then + echo "Usage: $0 " >&2 exit 1 fi @@ -27,7 +27,7 @@ if [[ ! -r "${site_config}" ]]; then exit 1 fi -# Site config provides site-specific ingestion settings. +# Site config provides the machine identity and ingestion settings. source "${site_config}" # Every site uses the standard deployment layout. The scheduler supplies its @@ -37,28 +37,31 @@ SIMBOARD_WORKDIR="${SIMBOARD_ROOT}/operations" SIMBOARD_MODULES="${SIMBOARD_ROOT}/repository/simboard/backend" SIMBOARD_RAW_LOG_DIR="${SIMBOARD_WORKDIR}/raw_logs" -: "${SIMBOARD_INGESTOR_MODULE:?SIMBOARD_INGESTOR_MODULE must be set by the site configuration}" - # ============================================================================= -# Run controls +# Operation configuration # ============================================================================= -# The command selects the scan mode. Site configuration supplies the archive -# lower bound; scheduler settings may narrow the archive scan or cap submissions. -export SCAN_MODE="${scan_mode}" - -if [[ $scan_mode == "archive" ]]; then - export ARCHIVE_YEAR_START="${ARCHIVE_YEAR_START:-${SIMBOARD_DEFAULT_ARCHIVE_YEAR_START:?SIMBOARD_DEFAULT_ARCHIVE_YEAR_START must be set by the site configuration}}" +if [[ $scan_mode == "diagnostics" ]]; then + module="app.scripts.ingestion.diagnostics_link_scanner" + log_prefix="simboard-diagnostics" + task_label="diagnostics" + module_label="scanner" +else + : "${SIMBOARD_INGESTOR_MODULE:?SIMBOARD_INGESTOR_MODULE must be set by the site configuration}" + module="${SIMBOARD_INGESTOR_MODULE}" + log_prefix="simboard-ingestion-${scan_mode}" + task_label="ingestion" + module_label="ingestor" + export SCAN_MODE="${scan_mode}" + export MAX_CASES_PER_RUN="${MAX_CASES_PER_RUN:-}" + if [[ $scan_mode == "archive" ]]; then + export ARCHIVE_YEAR_START="${ARCHIVE_YEAR_START:-${SIMBOARD_DEFAULT_ARCHIVE_YEAR_START:?SIMBOARD_DEFAULT_ARCHIVE_YEAR_START must be set by the site configuration}}" + fi fi -export MAX_CASES_PER_RUN="${MAX_CASES_PER_RUN:-}" - # ============================================================================= # API configuration # ============================================================================= -# Dry runs default to read-only remote-state validation. Set -# DRY_RUN_USE_REMOTE_STATE=false for credential-free offline scanning. -# Read both controls with safe defaults, then trim leading and trailing -# whitespace so scheduler values such as " false " are handled correctly below. +# Trim scheduler values before deciding whether API access is needed. dry_run_normalized="${DRY_RUN:-true}" dry_run_normalized="${dry_run_normalized#"${dry_run_normalized%%[![:space:]]*}"}" dry_run_normalized="${dry_run_normalized%"${dry_run_normalized##*[![:space:]]}"}" @@ -74,17 +77,35 @@ load_api_configuration() { } shopt -s nocasematch -case "${dry_run_normalized}" in - 0|false|no|off) - load_api_configuration - ;; - *) - case "${remote_state_normalized}" in - 0|false|no|off) ;; - *) load_api_configuration ;; - esac - ;; -esac +if [[ $scan_mode == "diagnostics" ]]; then + # Diagnostics dry runs are offline; export a canonical boolean for Python. + case "${dry_run_normalized}" in + 0|false|no|off) + load_api_configuration + export DRY_RUN=false + ;; + 1|true|yes|on) + export DRY_RUN=true + ;; + *) + echo "DRY_RUN must be a boolean" >&2 + exit 1 + ;; + esac +else + # Ingestion dry runs read remote state unless explicitly configured offline. + case "${dry_run_normalized}" in + 0|false|no|off) + load_api_configuration + ;; + *) + case "${remote_state_normalized}" in + 0|false|no|off) ;; + *) load_api_configuration ;; + esac + ;; + esac +fi shopt -u nocasematch # ============================================================================= @@ -109,17 +130,22 @@ ts="$(date -u +%Y%m%d_%H%M%S)" environment_lock_name="${SIMBOARD_ENV_FILE:-offline}" environment_lock_name="${environment_lock_name##*/}" mkdir -p -m 750 "${SIMBOARD_RAW_LOG_DIR}" -LOG_FILE="${SIMBOARD_RAW_LOG_DIR}/simboard-ingestion-${scan_mode}-${site}-${environment_lock_name}-${ts}.log" +LOG_FILE="${SIMBOARD_RAW_LOG_DIR}/${log_prefix}-${site}-${environment_lock_name}-${ts}.log" printf '[%s] launcher started: site=%s scan_mode=%s dry_run=%s\n' \ "$(date -Is)" "${site}" "${scan_mode}" "${dry_run_normalized}" >> "${LOG_FILE}" -printf '[%s] launcher configuration: site_config=%s dry_run_use_remote_state=%s ingestor_module=%s\n' \ - "$(date -Is)" "${site_config}" "${remote_state_normalized}" "${SIMBOARD_INGESTOR_MODULE}" >> "${LOG_FILE}" +if [[ $scan_mode == "diagnostics" ]]; then + printf '[%s] launcher configuration: site_config=%s scanner_module=%s\n' \ + "$(date -Is)" "${site_config}" "${module}" >> "${LOG_FILE}" +else + printf '[%s] launcher configuration: site_config=%s dry_run_use_remote_state=%s ingestor_module=%s\n' \ + "$(date -Is)" "${site_config}" "${remote_state_normalized}" "${module}" >> "${LOG_FILE}" +fi -LOCK_FILE="$SIMBOARD_WORKDIR/simboard-ingestion-${scan_mode}-${site}-${environment_lock_name}.lock" +LOCK_FILE="$SIMBOARD_WORKDIR/${log_prefix}-${site}-${environment_lock_name}.lock" exec 200>"$LOCK_FILE" if ! flock -n 200; then - echo "[$(date -Is)] lock already held; ingestion was not started, pid $$" >> "$LOG_FILE" - if [[ "${scan_mode}" == "archive" ]]; then + echo "[$(date -Is)] lock already held; ${task_label} was not started, pid $$" >> "$LOG_FILE" + if [[ "${scan_mode}" != "staging" ]]; then exit 1 fi exit 0 @@ -128,7 +154,7 @@ fi # Existing deployments leave their legacy lock files behind. Hold one when it # exists so a newly deployed launcher cannot overlap an in-flight old launcher. legacy_lock_file="$SIMBOARD_WORKDIR/SBCS-${site}-${environment_lock_name}.lock" -if [[ -e "${legacy_lock_file}" ]]; then +if [[ $scan_mode != "diagnostics" && -e "${legacy_lock_file}" ]]; then exec 201>"$legacy_lock_file" if ! flock -n 201; then echo "[$(date -Is)] legacy lock already held; ingestion was not started, pid $$" >> "$LOG_FILE" @@ -146,10 +172,10 @@ cleanup() { trap cleanup EXIT # ============================================================================= -# Ingestion execution +# Module execution # ============================================================================= -# Run the selected ingestor and append its structured events to this invocation's log. +# Run the selected module and append its structured events to this invocation's log. cd "${SIMBOARD_MODULES}" -printf '[%s] invoking ingestor: module=%s\n' \ - "$(date -Is)" "${SIMBOARD_INGESTOR_MODULE}" >> "${LOG_FILE}" -"${PYTHON_BIN}" -m "${SIMBOARD_INGESTOR_MODULE}" >> "$LOG_FILE" 2>&1 +printf '[%s] invoking %s: module=%s\n' \ + "$(date -Is)" "${module_label}" "${module}" >> "${LOG_FILE}" +"${PYTHON_BIN}" -m "${module}" >> "$LOG_FILE" 2>&1 diff --git a/backend/app/scripts/ingestion/sites/templates/crontab.example b/backend/app/scripts/ingestion/sites/templates/crontab.example index 1f4c377c..62ab82bc 100644 --- a/backend/app/scripts/ingestion/sites/templates/crontab.example +++ b/backend/app/scripts/ingestion/sites/templates/crontab.example @@ -63,3 +63,16 @@ SIMBOARD_ROOT=/path/to/simboard_root 0 12 * * * SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.dev.sh" "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis archive # Production 0 12 * * * SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.prod.sh" "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis archive + +# ============================================================================= +# Diagnostics scans +# ============================================================================= +# Link diagnostics daily at 14:00 UTC with independent locks and API credentials. +# DRY_RUN=false enables linking regardless of ingestion's global run mode. +# Use true for an optional offline scan. Archive settings are in +# diagnostics_archives.py; ingestion year/case limits do not apply. + +# Development +0 14 * * * DRY_RUN=false SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.dev.sh" "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis diagnostics +# Production +0 14 * * * DRY_RUN=false SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.prod.sh" "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis diagnostics diff --git a/backend/tests/features/ingestion/test_diagnostics_link_scanner.py b/backend/tests/features/ingestion/test_diagnostics_link_scanner.py index cc2e1b5e..381a2130 100644 --- a/backend/tests/features/ingestion/test_diagnostics_link_scanner.py +++ b/backend/tests/features/ingestion/test_diagnostics_link_scanner.py @@ -47,8 +47,7 @@ def resolve_archive(_machine: str) -> DiagnosticsArchive: else: monkeypatch.setenv("MACHINE_NAME", machine_name) - with pytest.raises(ValueError, match="MACHINE_NAME is required"): - run() + assert run() == 1 def _case( @@ -403,10 +402,13 @@ def test_run_submits_exact_payload_and_bearer_auth( ) in events +@pytest.mark.parametrize("filtered", [False, True]) def test_dry_run_requires_no_api_configuration( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, filtered: bool ) -> None: _case(tmp_path, "production/type/case") + if filtered: + _case(tmp_path, "production/type/excluded") monkeypatch.setattr( "app.scripts.ingestion.diagnostics_link_scanner._resolve_archive", lambda _machine: DiagnosticsArchive(str(tmp_path), BASE_URL), @@ -420,7 +422,8 @@ def test_dry_run_requires_no_api_configuration( "app.scripts.ingestion.diagnostics_link_scanner._log_event", lambda event, fields=None: events.append((event, fields)), ) - assert run() == 0 + included_case_paths = {Path("production/type/case")} if filtered else None + assert run(included_case_paths=included_case_paths) == 0 candidate_events = [ fields for event, fields in events @@ -444,6 +447,7 @@ def test_dry_run_requires_no_api_configuration( "deferred_state_lookups": 0, "submitted_links": 0, "failed_link_submissions": 0, + "outcome": "completed", }, ) @@ -483,3 +487,185 @@ def test_run_defers_after_exhausted_state_lookup( "diagnostics_scanner_state_lookup_deferred", {"archive_relative_case_path": "production/type/case", "status_code": 503}, ) in events + assert events[-1][0] == "diagnostics_scanner_completed" + assert events[-1][1]["outcome"] == "failed" + + +@pytest.mark.parametrize("dry_run", [" TRUE ", "Yes", "ON", "1"]) +def test_dry_run_boolean_variants_never_open_api_client( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, dry_run: str +) -> None: + monkeypatch.setenv("MACHINE_NAME", "perlmutter") + monkeypatch.setenv("DRY_RUN", dry_run) + monkeypatch.setattr( + "app.scripts.ingestion.diagnostics_link_scanner._resolve_archive", + lambda _machine: DiagnosticsArchive(str(tmp_path), BASE_URL), + ) + monkeypatch.setattr( + "app.scripts.ingestion.diagnostics_link_scanner.httpx.Client", + lambda **_kwargs: pytest.fail("dry run must not open an API client"), + ) + assert run() == 0 + + +@pytest.mark.parametrize( + "failure_phase", ["configuration", "discovery", "state_lookup", "link_submission"] +) +def test_fatal_failure_reports_safe_context_and_partial_summary( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, failure_phase: str +) -> None: + from app.scripts.ingestion import diagnostics_link_scanner as scanner + + _case(tmp_path, "production/type/first") + _case(tmp_path, "production/type/second") + events: list[tuple[str, dict]] = [] + monkeypatch.setattr( + scanner, "_log_event", lambda event, fields: events.append((event, fields)) + ) + monkeypatch.setenv("MACHINE_NAME", "perlmutter") + monkeypatch.setenv("DRY_RUN", "false") + monkeypatch.setenv("SIMBOARD_API_BASE_URL", "https://api.example.org") + monkeypatch.setenv("SIMBOARD_API_TOKEN", "secret-token") + monkeypatch.setattr( + scanner, + "_resolve_archive", + lambda _: DiagnosticsArchive(str(tmp_path), BASE_URL), + ) + + def fail(*_args, **_kwargs): + raise RuntimeError("sensitive-response secret-token") + + if failure_phase == "configuration": + monkeypatch.setattr(scanner, "_resolve_archive", fail) + elif failure_phase == "discovery": + monkeypatch.setattr(scanner, "_discover", fail) + else: + client = _Client(httpx.Response(200, json=None)) + + def get(*_args, **_kwargs): + if client.post_calls and failure_phase == "state_lookup": + fail() + return httpx.Response(200, json=None) + + monkeypatch.setattr(client, "get", get) + if failure_phase == "link_submission": + monkeypatch.setattr(client, "post", fail) + monkeypatch.setattr(scanner.httpx, "Client", lambda **_kwargs: client) + + assert run() == 1 + failures = [ + fields for event, fields in events if event == "diagnostics_scanner_failed" + ] + assert len(failures) == 1 + assert failures[0]["phase"] == failure_phase + assert failures[0]["error_type"] == "RuntimeError" + assert failures[0]["traceback_frames"] + assert "secret-token" not in str(events) + assert "sensitive-response" not in str(events) + summaries = [ + fields for event, fields in events if event == "diagnostics_scanner_completed" + ] + assert len(summaries) == 1 + assert summaries[0]["outcome"] == "failed" + assert summaries[0]["submitted_links"] == ( + 1 if failure_phase == "state_lookup" else 0 + ) + assert summaries[0]["discovered_candidates"] == ( + 2 if failure_phase in {"state_lookup", "link_submission"} else 0 + ) + if failure_phase in {"state_lookup", "link_submission"}: + assert failures[0]["archive_relative_case_path"].startswith("production/type/") + + +@pytest.mark.parametrize( + ("dry_run", "api_base", "token", "error_type"), + [ + pytest.param( + "typo", + "https://api.example.org", + "secret-token", + "ValueError", + id="invalid-dry-run", + ), + pytest.param( + "false", "https://api.example.org", None, "KeyError", id="missing-token" + ), + pytest.param( + "false", "https://api.example.org", "", "ValueError", id="empty-token" + ), + pytest.param("false", None, "secret-token", "KeyError", id="missing-api-url"), + pytest.param("false", "", "secret-token", "ValueError", id="empty-api-url"), + ], +) +def test_invalid_configuration_fails_before_discovery( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + dry_run: str, + api_base: str | None, + token: str | None, + error_type: str, +) -> None: + from app.scripts.ingestion import diagnostics_link_scanner as scanner + + events: list[tuple[str, dict]] = [] + monkeypatch.setenv("MACHINE_NAME", "perlmutter") + monkeypatch.setenv("DRY_RUN", dry_run) + for name, value in ( + ("SIMBOARD_API_BASE_URL", api_base), + ("SIMBOARD_API_TOKEN", token), + ): + if value is None: + monkeypatch.delenv(name, raising=False) + else: + monkeypatch.setenv(name, value) + monkeypatch.setattr( + scanner, "_log_event", lambda event, fields: events.append((event, fields)) + ) + monkeypatch.setattr( + scanner, + "_resolve_archive", + lambda _: DiagnosticsArchive(str(tmp_path), BASE_URL), + ) + monkeypatch.setattr( + scanner, "_discover", lambda *_args: pytest.fail("must validate first") + ) + assert run() == 1 + failure = next( + fields for event, fields in events if event == "diagnostics_scanner_failed" + ) + assert failure["phase"] == "configuration" + assert failure["error_type"] == error_type + assert events[-1][1]["outcome"] == "failed" + assert "secret-token" not in str(events) + + +@pytest.mark.parametrize("body", [b"not-json", b'["invalid-state"]']) +def test_invalid_state_response_emits_failed_summary( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, body: bytes +) -> None: + from app.scripts.ingestion import diagnostics_link_scanner as scanner + + _case(tmp_path, "production/type/case") + client = _Client(httpx.Response(200, content=body)) + events: list[tuple[str, dict]] = [] + monkeypatch.setattr( + scanner, "_log_event", lambda event, fields: events.append((event, fields)) + ) + monkeypatch.setattr( + scanner, + "_resolve_archive", + lambda _: DiagnosticsArchive(str(tmp_path), BASE_URL), + ) + monkeypatch.setattr(scanner.httpx, "Client", lambda **_kwargs: client) + monkeypatch.setenv("MACHINE_NAME", "perlmutter") + monkeypatch.setenv("DRY_RUN", "false") + monkeypatch.setenv("SIMBOARD_API_BASE_URL", "https://api.example.org") + monkeypatch.setenv("SIMBOARD_API_TOKEN", "token") + assert run() == 1 + failure = next( + fields for event, fields in events if event == "diagnostics_scanner_failed" + ) + assert failure["phase"] == "state_lookup" + assert failure["archive_relative_case_path"] == "production/type/case" + assert events[-1][1]["outcome"] == "failed" + assert client.post_calls == [] diff --git a/backend/tests/features/ingestion/test_site_collection_launcher.py b/backend/tests/features/ingestion/test_site_collection_launcher.py index 67c4ef95..3728ed06 100644 --- a/backend/tests/features/ingestion/test_site_collection_launcher.py +++ b/backend/tests/features/ingestion/test_site_collection_launcher.py @@ -107,7 +107,7 @@ def test_launcher_runs_configured_ingestor_offline(tmp_path: Path) -> None: @pytest.mark.parametrize( ("scan_mode", "expected_returncode"), - [("staging", 0), ("archive", 1)], + [("staging", 0), ("archive", 1), ("diagnostics", 1)], ) def test_launcher_reports_mode_specific_lock_contention( tmp_path: Path, scan_mode: str, expected_returncode: int @@ -162,15 +162,127 @@ def test_launcher_reports_mode_specific_lock_contention( assert result.returncode == expected_returncode assert flock_capture_path.read_text(encoding="utf-8").splitlines() == ["-n 200"] - raw_logs = list( - (work_dir / "raw_logs").glob( - f"simboard-ingestion-{scan_mode}-test-offline-*.log" - ) + prefix = ( + "simboard-diagnostics" + if scan_mode == "diagnostics" + else f"simboard-ingestion-{scan_mode}" ) + raw_logs = list((work_dir / "raw_logs").glob(f"{prefix}-test-offline-*.log")) assert len(raw_logs) == 1 - assert "lock already held; ingestion was not started" in raw_logs[0].read_text( - encoding="utf-8" + assert "lock already held;" in raw_logs[0].read_text(encoding="utf-8") + + +@pytest.mark.parametrize("environment", ["dev", "prod"]) +@pytest.mark.parametrize("dry_run", [None, " TRUE ", "On", " false ", "OFF"]) +def test_diagnostics_launcher_dispatch_and_environment( + tmp_path: Path, environment: str, dry_run: str | None +) -> None: + root = tmp_path / "deployment" + operations = root / "operations" + operations.mkdir(parents=True) + (root / "repository/simboard/backend/.venv").mkdir(parents=True) + bin_dir = tmp_path / "bin" + bin_dir.mkdir() + capture = tmp_path / "capture" + fake_python = _write_executable( + bin_dir / "python", + "#!/usr/bin/env bash\n" + 'printf "%s\\n" "$*" "$MACHINE_NAME" "$DRY_RUN" ' + '"${SIMBOARD_API_BASE_URL-unset}" "${SIMBOARD_API_TOKEN-unset}" ' + '"${SCAN_MODE-unset}" "${ARCHIVE_YEAR_START-unset}" > "$CAPTURE_PATH"\n' + 'exit "${PYTHON_EXIT_CODE:-0}"\n', ) + # A diagnostics run must not acquire the legacy ingestion lock. + (operations / f"SBCS-chrysalis-env.{environment}.sh.lock").touch() + _write_executable( + bin_dir / "flock", '#!/usr/bin/env bash\n[[ "$*" == "-n 200" ]]\n' + ) + config = tmp_path / "site.config" + config.write_text( + f"export MACHINE_NAME=chrysalis\nexport PYTHON_BIN={shlex.quote(str(fake_python))}\n", + encoding="utf-8", + ) + live = dry_run is not None and dry_run.strip().lower() in {"false", "off"} + api_file = operations / f"env.{environment}.sh" + if live: + api_file.write_text( + f"export SIMBOARD_API_BASE_URL=https://{environment}.example.test\n" + f"export SIMBOARD_API_TOKEN={environment}-token\n", + encoding="utf-8", + ) + # For dry runs the selected API file intentionally does not exist. + env = os.environ.copy() + for key in ( + "DRY_RUN", + "SIMBOARD_API_BASE_URL", + "SIMBOARD_API_TOKEN", + "SIMBOARD_INGESTOR_MODULE", + "SCAN_MODE", + "ARCHIVE_YEAR_START", + ): + env.pop(key, None) + env.update( + SIMBOARD_ROOT=str(root), + SIMBOARD_SITE_CONFIG=str(config), + SIMBOARD_ENV_FILE=str(api_file), + CAPTURE_PATH=str(capture), + PATH=f"{bin_dir}:{env['PATH']}", + ) + if dry_run is not None: + env["DRY_RUN"] = dry_run + + def invoke(): + return subprocess.run( + [_launcher_path(), "chrysalis", "diagnostics"], + capture_output=True, + check=False, + env=env, + text=True, + ) + + result = invoke() + assert result.returncode == 0, result.stderr + assert capture.read_text().splitlines() == [ + "-m app.scripts.ingestion.diagnostics_link_scanner", + "chrysalis", + "false" if live else "true", + f"https://{environment}.example.test" if live else "unset", + f"{environment}-token" if live else "unset", + "unset", + "unset", + ] + assert ( + operations / f"simboard-diagnostics-chrysalis-env.{environment}.sh.lock" + ).exists() + logs = list((operations / "raw_logs").glob("simboard-diagnostics-*.log")) + assert logs + assert f"{environment}-token" not in logs[0].read_text() + assert "dry_run_use_remote_state=" not in logs[0].read_text() + assert "invoking scanner:" in logs[0].read_text() + + env["PYTHON_EXIT_CODE"] = "7" + assert invoke().returncode == 7 + if live: + api_file.write_text("export SIMBOARD_API_BASE_URL=https://example.test\n") + result = invoke() + assert result.returncode != 0 + assert "SIMBOARD_API_TOKEN failed to be set" in result.stderr + env["DRY_RUN"] = "typo" + result = invoke() + assert result.returncode != 0 + assert "DRY_RUN must be a boolean" in result.stderr + + +def test_cron_template_schedules_diagnostics_independently() -> None: + template = (_launcher_path().parent / "templates/crontab.example").read_text() + jobs = [ + line for line in template.splitlines() if line.endswith("chrysalis diagnostics") + ] + assert len(jobs) == 2 + for job, environment in zip(jobs, ("dev", "prod"), strict=True): + assert job.startswith("0 14 * * * DRY_RUN=false ") + assert f"/operations/env.{environment}.sh" in job + assert "&&" not in job @pytest.mark.parametrize( diff --git a/docs/operations/nersc-spin-runbook.md b/docs/operations/nersc-spin-runbook.md index b509e92d..cfe2c701 100644 --- a/docs/operations/nersc-spin-runbook.md +++ b/docs/operations/nersc-spin-runbook.md @@ -566,7 +566,113 @@ Security context requirements for NERSC global file system (NGF/CFS) mounts: Source: [NERSC Spin Storage - NERSC Global File Systems](https://docs.nersc.gov/services/spin/storage/#nersc-global-file-systems). -### Workload 4: Frontend Deployment (`frontend`) +### Workload 4: NERSC Diagnostics Scanner CronJob + +Create this CronJob in Rancher for each target API namespace. It links published +diagnostics to existing cases independently of the ingestion CronJobs. Run the +Python scanner directly, not the host-side launcher. + +#### Setup procedure + +Under **Storage -> Secrets**, create the **Opaque** secret +`nersc-diagnostics-scanner-env`: + +| Key | Value | +| --- | --- | +| `MACHINE_NAME` | `perlmutter` | +| `DRY_RUN` | `false` | +| `SIMBOARD_API_BASE_URL` | `http://backend:8000` | +| `SIMBOARD_API_TOKEN` | Ingestion service-account token authorized for diagnostics | + +Use the same namespace's backend image and API credentials. Keep `DRY_RUN=false` +for linking; `true` is an optional offline scan. Omitting it defaults to dry run. + +Open **Workloads -> CronJobs -> Create**: + +| Rancher field | Value | +| --- | --- | +| Name | `nersc-diagnostics-scanner` | +| Schedule | `0 14 * * *` | +| Time zone | `Etc/UTC` | + +#### 1. CronJob tab + +`Scaling and Upgrade Policy`: + +| Rancher field | Value | +| --- | --- | +| Concurrency policy | `Skip next run if current run hasn't finished` (`Forbid`) | +| Successful jobs history limit | `3` | +| Failed jobs history limit | `3` | +| Job backoff limit | `0` (retry at the next schedule) | + +The daily scan is offset from 12:00 archive ingestion. If the time-zone field is +unavailable, verify the controller uses UTC. Avoid overlapping manual jobs; +`Forbid` only controls scheduled jobs. + +#### 2. Pod tab + +`Security Context` and `Pod`: + +| Rancher field | Value | +| --- | --- | +| Pod Filesystem Group | `62756` | +| Restart policy | `Never` | + +`Storage`: + +| Rancher field | Value | +| --- | --- | +| Volume type | `Bind-Mount` | +| Volume name | `diagnostics` | +| Path on node | `/global/cfs/cdirs/e3sm/www/diagnostics_archive` | +| The Path on the Node must be | `An existing directory` | + +#### 3. Container tab + +`General`: + +| Rancher field | Value | +| --- | --- | +| Container Name | `nersc-diagnostics-scanner` | +| Container image | `registry.nersc.gov/e3sm/simboard/backend:` | +| Pull policy | `Always` for `:dev`; `IfNotPresent` for versioned tags | +| Image pull secret | `registry-nersc` | +| Working directory | `/app` | +| Command | `python` | +| Arguments | `-m app.scripts.ingestion.diagnostics_link_scanner` | +| Environment Variables | Type: Secret, Secret: `nersc-diagnostics-scanner-env` | + +`Security Context`: + +| Rancher field | Value | +| --- | --- | +| Run as User | Numeric NERSC UID authorized to read the archive | +| allowPrivilegeEscalation | `false` | +| privileged | `false` | +| capabilities | drop `ALL` | + +`Storage`: + +| Rancher field | Value | +| --- | --- | +| Archive volume | `diagnostics` | +| Archive mount path | `/global/cfs/cdirs/e3sm/www/diagnostics_archive` | +| Archive read only | `true` | + +Mount the archive at the exact path registered in `diagnostics_archives.py` and +confirm the configured UID/group can read it. No root/URL environment override +is needed. + +#### Verify live linking + +1. Trigger a one-off job and confirm startup logs show `perlmutter`, the correct + archive root and `dry_run=false`. +2. Check the [scanner results](setup-ingestion-operations.md#check-scanner-results) + and verify links in SimBoard. Handled API failures can occur even when the pod succeeds. +3. Confirm the next scheduled run and monitor duration: every scan walks the full archive. + +### Workload 5: Frontend Deployment (`frontend`) Workloads -> Deployments -> Create (top-right) @@ -676,6 +782,9 @@ Ingress annotation if that backend limit changes. 6. Update/redeploy frontend deployment with the target frontend image tag, then verify frontend pod status. 7. Create/confirm an admin account (Rancher pod shell), then provision ingestion service-account token and create/update secrets `nersc-staging-ingestor-env` and `nersc-archive-ingestor-env`. 8. Create/update CronJobs `nersc-staging-ingestor` and `nersc-archive-ingestor`, run one-off dry runs (`DRY_RUN=true`) for both, then set `DRY_RUN=false`. + Separately create `nersc-diagnostics-scanner-env` and `nersc-diagnostics-scanner` + using Workload 4 with `DRY_RUN=false`, verify its archive mount and live linking. + Keep existing ingestion workloads unchanged. 9. Verify ingress routing under **Service Discovery → Ingresses** for `lb` and confirm both frontend and backend hosts resolve via HTTPS. ## Failure Handling diff --git a/docs/operations/setup-ingestion-operations.md b/docs/operations/setup-ingestion-operations.md index 169bb467..e9e33ce9 100644 --- a/docs/operations/setup-ingestion-operations.md +++ b/docs/operations/setup-ingestion-operations.md @@ -1,7 +1,8 @@ # Set Up Ingestion Operations Use this guide to provision and operate scheduled performance ingestion, the v3 -backfill, and diagnostics discovery. Start every new job with `DRY_RUN=true`. +backfill, and diagnostics discovery. Start new ingestion jobs with `DRY_RUN=true`; +scheduled diagnostics jobs are configured for live linking. ## Prerequisites @@ -139,11 +140,15 @@ The copied crontab: default, so local refreshes proceed without that lock; - adds the standard user-level `uv` location to the refresh command's `PATH`; adjust it if the scheduler account installs `uv` elsewhere; -- creates staging and archive jobs for both development and production; and +- creates staging, archive, and independent diagnostics jobs for both development and production; - gives each ingestion command its own `SIMBOARD_ENV_FILE`; development and production jobs use separate locks and may run at the same time; staging and archive jobs for one environment also use separate locks and may run at the - same time. + same time; and +- runs diagnostics daily at 14:00 UTC with separate locks and live linking. + +Diagnostics entries use `DRY_RUN=false` independently of ingestion. Supply valid +API credentials before installing them; use `DRY_RUN=true` for an optional dry scan. ### Reference: configuration files and variables @@ -166,8 +171,10 @@ Do not put tokens in the site config or crontab. `DRY_RUN_USE_REMOTE_STATE=false 1. Run `make operations-provision` to add `raw_logs/` and `quality_assurance/` without replacing existing deployment-local files. 2. Update the copied crontab with `make operations-init-cron` only after - preserving and manually updating its schedules and `SIMBOARD_ROOT`, or edit - the existing copied crontab to use the new provisioning-log path. + backing up the existing file; the initializer refuses to overwrite it. + Alternatively, merge the template's diagnostics entries and provisioning-log + path manually, preserving local schedules and `SIMBOARD_ROOT`. Diagnostics + entries enable live linking, so confirm API credentials before installation. 3. Let existing `SBCS-*` lock files remain until no scheduler invocation uses the previous launcher or refresh helper. New helpers acquire an existing legacy lock as a compatibility guard, so they do not overlap a running old @@ -245,29 +252,49 @@ addition to `SIMBOARD_ROOT` and `env`. Optional variables in the v3 file are `OLD_PERF_ARCHIVE_ROOT`, `MAX_ATTEMPTS`, `MAX_CASES_PER_RUN`, `REQUEST_TIMEOUT_SECONDS`, and `ARCHIVE_YEAR_END`. -## Diagnostics discovery operation +## Diagnostics linking + +The scanner links published zppy diagnostics to existing SimBoard cases. Publish +output using [Configure zppy Diagnostics for SimBoard](../user/diagnostics.md) +and verify the machine's archive in `backend/app/scripts/ingestion/diagnostics_archives.py`. -Diagnostics discovery is separate from performance ingestion and links published zppy output to an existing SimBoard case. +### Scheduled scans on Chrysalis -1. Configure and publish zppy output as described in [Configure zppy Diagnostics for SimBoard](../user/diagnostics.md). -2. Add the machine's filesystem root and public URL to the reviewed registry in `backend/app/scripts/ingestion/diagnostics_archives.py`. -3. Run a dry scan from the backend directory: +1. Confirm the scheduler account can read the diagnostics archive and the + protected API files contain authorized service-account credentials. +2. Merge the template's dev/prod diagnostics entries into the installed crontab. + They run daily at 14:00 UTC with `DRY_RUN=false` and separate locks. +3. To run a scan manually, select the API environment: ```bash - MACHINE_NAME=perlmutter DRY_RUN=true \ - uv run python -m app.scripts.ingestion.diagnostics_link_scanner + DRY_RUN=false SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.prod.sh" \ + "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis diagnostics ``` -4. For a live scan, set `DRY_RUN=false` and provide `SIMBOARD_API_BASE_URL` and `SIMBOARD_API_TOKEN` in the scheduler environment. + Use `env.dev.sh` for development or `DRY_RUN=true` for an offline dry scan. +4. Review `operations/raw_logs/simboard-diagnostics-chrysalis-*.log` and verify + links in SimBoard. Each scan walks the full archive; ingestion year/case limits + do not apply. -| Configure in | Variables | -| --- | --- | -| Scheduler environment | `MACHINE_NAME`, `DRY_RUN` | -| Live-scan scheduler secret | `SIMBOARD_API_BASE_URL`, `SIMBOARD_API_TOKEN` | +### Scheduled scans on NERSC Spin + +Configure a separate CronJob through Rancher using the +[NERSC Spin runbook](nersc-spin-runbook.md#workload-4-nersc-diagnostics-scanner-cronjob). + +### Check scanner results + +- `diagnostics_scanner_startup_configuration` reports the machine, archive and run mode. +- `diagnostics_scanner_completed` reports the outcome and counters collected so far. +- Fatal errors log `diagnostics_scanner_failed` and exit nonzero. Discovery failures + may leave the candidate count at zero; abrupt termination may omit the summary. +- Check `deferred_state_lookups` and `failed_link_submissions` even when the run + succeeds: handled API failures are counted without a nonzero exit. Later scans + retry deferred work and skip unchanged links. ## Operate safely -- Review dry-run logs before enabling a live job. +- Review ingestion dry-run logs before enabling live ingestion. Diagnostics dry + runs are optional; review live diagnostics counters and links after deployment. - A successful validation is not an ingestion; only a successful request records an execution as processed. - Correct filesystem or network failures and let the next scheduled run retry them. - Keep tokens out of logs, source control, site configs, and crontabs. diff --git a/docs/operations/test-ingestion-operations.md b/docs/operations/test-ingestion-operations.md index 0b6e6bd8..70968600 100644 --- a/docs/operations/test-ingestion-operations.md +++ b/docs/operations/test-ingestion-operations.md @@ -100,6 +100,22 @@ grep -n "SIMBOARD_ROOT\|refresh_repository\|site_ingestion_launcher" \ Do **not** run `crontab "${SIMBOARD_ROOT}/operations/chrysalis.crontab"` during this test. +Confirm two `chrysalis diagnostics` entries at `0 14 * * *`, using `env.dev.sh` +and `env.prod.sh` with `DRY_RUN=false`. Do not execute these live jobs with dummy +credentials or without the site archive. Staging/archive schedules are unchanged. + +Run the isolated launcher/scanner tests from the repository root: + +```bash +uv run --project backend pytest \ + backend/tests/features/ingestion/test_site_collection_launcher.py \ + backend/tests/features/ingestion/test_diagnostics_link_scanner.py \ + --noconftest --no-cov +``` + +These tests use temporary archives and mocked runtimes/API calls. +`--noconftest` skips the unused application-wide database setup. + ## 5. Test a no-op refresh ```bash From 568be43b4995310d8e3e94874707f2a972d87048 Mon Sep 17 00:00:00 2001 From: Tom Vo Date: Tue, 6 Oct 2026 12:25:15 -0500 Subject: [PATCH 2/2] [DevOps]: Add Make targets for manual diagnostics scans --- Makefile | 30 +++ .../test_diagnostics_make_targets.py | 177 ++++++++++++++++++ docs/operations/nersc-spin-runbook.md | 3 +- docs/operations/setup-ingestion-operations.md | 22 ++- docs/operations/test-ingestion-operations.md | 5 +- 5 files changed, 228 insertions(+), 9 deletions(-) create mode 100644 backend/tests/features/ingestion/test_diagnostics_make_targets.py diff --git a/Makefile b/Makefile index fd91ba62..9b5cec93 100644 --- a/Makefile +++ b/Makefile @@ -56,6 +56,8 @@ help: @echo " make v3-diagnostics-dry-run SIMBOARD_ROOT= env= # Reconcile Chrysalis v3 diagnostics without writes" @echo " make v3-diagnostics-apply SIMBOARD_ROOT= env= # Backfill and link Chrysalis v3 diagnostics" @echo " Optional: case_name= to retry one v3 diagnostics case" + @echo " make diagnostics-dry-run SIMBOARD_ROOT= site= env= # Discover published diagnostics offline" + @echo " make diagnostics-apply SIMBOARD_ROOT= site= env= # Link published diagnostics to existing cases" @echo "" @echo "Frontend:" @@ -186,6 +188,34 @@ V3_ENV_INITIALIZER := $(INGESTION_OPERATIONS_DIR)/initialize_v3_environment.sh V3_ENV_TEMPLATE := $(BACKEND_DIR)/app/scripts/ingestion/v3_data/lcrc-v3.env.example INGESTION_PROVISION_SCRIPT := $(INGESTION_OPERATIONS_DIR)/provision_operations.sh INGESTION_REFRESH_SCRIPT := $(INGESTION_OPERATIONS_DIR)/refresh_repository.sh +DIAGNOSTICS_LAUNCHER := backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh + +.PHONY: diagnostics-dry-run diagnostics-apply + +# Use the host launcher for its protected API configuration, locks, and logs. +# Dry scans select an environment for log/lock naming but never load credentials. +diagnostics-dry-run diagnostics-apply: + @if [ -z "$(SIMBOARD_ROOT)" ] || [ -z "$(site)" ] || [ -z "$(env)" ]; then \ + echo "Usage: make $@ SIMBOARD_ROOT= site= env=" >&2; \ + exit 1; \ + fi; \ + case "$(site)" in *[!a-z0-9_-]*) echo "site must contain only lowercase letters, digits, underscores, or hyphens" >&2; exit 1 ;; esac; \ + if [ ! -r "$(INGESTION_SITE_CONFIGS_DIR)/$(site).config" ]; then \ + echo "Site configuration not readable: $(INGESTION_SITE_CONFIGS_DIR)/$(site).config" >&2; \ + exit 1; \ + fi; \ + if [ "$(env)" != "dev" ] && [ "$(env)" != "prod" ]; then \ + echo "env must be dev or prod" >&2; \ + exit 1; \ + fi; \ + SIMBOARD_ROOT="$(SIMBOARD_ROOT)" SIMBOARD_ENV_FILE="$(SIMBOARD_ROOT)/operations/env.$(env).sh" \ + SIMBOARD_SITE_CONFIG="$(CURDIR)/$(INGESTION_SITE_CONFIGS_DIR)/$(site).config" \ + DRY_RUN=$(if $(filter diagnostics-dry-run,$@),true,false) \ + bash "$(DIAGNOSTICS_LAUNCHER)" "$(site)" diagnostics || { \ + exit_code=$$?; \ + echo "Diagnostics launcher failed (exit $$exit_code). If a run log was created, check $(SIMBOARD_ROOT)/operations/raw_logs/simboard-diagnostics-$(site)-env.$(env).sh-*.log" >&2; \ + exit "$$exit_code"; \ + } operations-provision: @if [ -z "$(SIMBOARD_ROOT)" ]; then \ diff --git a/backend/tests/features/ingestion/test_diagnostics_make_targets.py b/backend/tests/features/ingestion/test_diagnostics_make_targets.py new file mode 100644 index 00000000..7dd7f0f3 --- /dev/null +++ b/backend/tests/features/ingestion/test_diagnostics_make_targets.py @@ -0,0 +1,177 @@ +"""Exercise operator Make targets through the host launcher without real scans.""" + +import os +import subprocess +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[4] + + +@pytest.fixture +def diagnostics_workspace(tmp_path: Path) -> tuple[Path, Path, dict[str, str]]: + # Include spaces to exercise quoting of deployment and interpreter paths. + root = tmp_path / "simboard root" + operations = root / "operations" + operations.mkdir(parents=True) + backend = root / "repository/simboard/backend" + backend.mkdir(parents=True) + (backend / ".venv").mkdir() + python = backend / "fake python" + python.write_text( + "#!/usr/bin/env bash\n" + 'printf "%s\\n" "$DRY_RUN" "$MACHINE_NAME" "$SIMBOARD_ENV_FILE" ' + '"${SIMBOARD_API_BASE_URL-unset}" "$*" > "$CAPTURE_PATH"\n' + 'exit "${SCANNER_EXIT_CODE:-0}"\n', + encoding="utf-8", + ) + python.chmod(0o755) + capture = tmp_path / "capture.txt" + process_env = os.environ.copy() + for key in ( + "SIMBOARD_ROOT", + "SIMBOARD_SITE_CONFIG", + "SIMBOARD_ENV_FILE", + "SIMBOARD_API_BASE_URL", + "SIMBOARD_API_TOKEN", + "MACHINE_NAME", + "MAKEFLAGS", + "MFLAGS", + "MAKEOVERRIDES", + ): + process_env.pop(key, None) + process_env.update(PYTHON_BIN=str(python), CAPTURE_PATH=str(capture)) + return root, capture, process_env + + +def _make(target: str, args: list[str], process_env: dict[str, str]): + return subprocess.run( + ["make", "--no-print-directory", target, *args], + cwd=REPO_ROOT, + env=process_env, + capture_output=True, + text=True, + check=False, + ) + + +@pytest.mark.parametrize("api_env", ["dev", "prod"]) +@pytest.mark.parametrize( + ("target", "dry_run"), + [("diagnostics-dry-run", "true"), ("diagnostics-apply", "false")], +) +def test_targets_select_mode_environment_and_launcher_logs( + diagnostics_workspace, target: str, dry_run: str, api_env: str +) -> None: + root, capture, process_env = diagnostics_workspace + # The explicit target mode must win over inherited or command-line DRY_RUN. + opposite_mode = "false" if dry_run == "true" else "true" + process_env["DRY_RUN"] = opposite_mode + conflicting_config = root / "unexpected.config" + conflicting_config.write_text('echo "must not load another site" >&2\nexit 99\n') + process_env["SIMBOARD_SITE_CONFIG"] = str(conflicting_config) + env_file = root / f"operations/env.{api_env}.sh" + if dry_run == "true": + env_file.write_text('echo "must not load dry-run credentials" >&2\nexit 99\n') + else: + env_file.write_text( + f"export SIMBOARD_API_BASE_URL=https://{api_env}.example.org\n" + "export SIMBOARD_API_TOKEN=test-token\n" + ) + result = _make( + target, + [ + f"SIMBOARD_ROOT={root}", + "site=chrysalis", + f"env={api_env}", + f"DRY_RUN={opposite_mode}", + ], + process_env, + ) + assert result.returncode == 0, result.stderr + assert capture.read_text().splitlines() == [ + dry_run, + "chrysalis", + str(env_file), + "unset" if dry_run == "true" else f"https://{api_env}.example.org", + "-m app.scripts.ingestion.diagnostics_link_scanner", + ] + logs = list((root / "operations/raw_logs").glob("simboard-diagnostics-*.log")) + assert len(logs) == 1 + assert f"chrysalis-env.{api_env}.sh-" in logs[0].name + assert "exit_code 0" in logs[0].read_text() + assert ( + root / f"operations/simboard-diagnostics-chrysalis-env.{api_env}.sh.lock" + ).exists() + assert "test-token" not in result.stdout + result.stderr + logs[0].read_text() + + +@pytest.mark.parametrize("target", ["diagnostics-dry-run", "diagnostics-apply"]) +@pytest.mark.parametrize( + ("missing", "site", "api_env", "message"), + [ + ("root", "chrysalis", "prod", "Usage:"), + ("site", "chrysalis", "prod", "Usage:"), + ("env", "chrysalis", "prod", "Usage:"), + (None, "unknown", "prod", "Site configuration not readable"), + (None, "../chrysalis", "prod", "site must contain only"), + (None, "Chrysalis", "prod", "site must contain only"), + (None, "chrysalis", "staging", "env must be dev or prod"), + (None, "chrysalis", "", "Usage:"), + ], +) +def test_targets_reject_invalid_arguments_before_execution( + diagnostics_workspace, target, missing, site, api_env, message +) -> None: + root, capture, process_env = diagnostics_workspace + args = { + "root": f"SIMBOARD_ROOT={root}", + "site": f"site={site}", + "env": f"env={api_env}", + } + result = _make( + target, [value for key, value in args.items() if key != missing], process_env + ) + assert result.returncode != 0 + assert message in result.stderr + assert not capture.exists() + assert not (root / "operations/raw_logs").exists() + + +@pytest.mark.parametrize("target", ["diagnostics-dry-run", "diagnostics-apply"]) +def test_targets_propagate_scanner_failure(diagnostics_workspace, target) -> None: + root, capture, process_env = diagnostics_workspace + (root / "operations/env.prod.sh").write_text( + "export SIMBOARD_API_BASE_URL=https://prod.example.org\n" + "export SIMBOARD_API_TOKEN=test-token\n" + ) + process_env["SCANNER_EXIT_CODE"] = "7" + result = _make( + target, [f"SIMBOARD_ROOT={root}", "site=chrysalis", "env=prod"], process_env + ) + assert result.returncode != 0 + assert capture.exists() + assert "Diagnostics launcher failed (exit 7)" in result.stderr + assert str(root / "operations/raw_logs") in result.stderr + log = next((root / "operations/raw_logs").glob("simboard-diagnostics-*.log")) + assert "exit_code 7" in log.read_text() + + +@pytest.mark.parametrize( + ("target", "succeeds"), + [("diagnostics-dry-run", True), ("diagnostics-apply", False)], +) +def test_only_live_scans_require_api_environment_file( + diagnostics_workspace, target, succeeds +) -> None: + root, capture, process_env = diagnostics_workspace + result = _make( + target, [f"SIMBOARD_ROOT={root}", "site=chrysalis", "env=prod"], process_env + ) + assert (result.returncode == 0) is succeeds + assert capture.exists() is succeeds + if not succeeds: + assert "env.prod.sh" in result.stderr + assert "Diagnostics launcher failed" in result.stderr + assert not (root / "operations/raw_logs").exists() diff --git a/docs/operations/nersc-spin-runbook.md b/docs/operations/nersc-spin-runbook.md index cfe2c701..8e21fa9b 100644 --- a/docs/operations/nersc-spin-runbook.md +++ b/docs/operations/nersc-spin-runbook.md @@ -669,7 +669,8 @@ is needed. 1. Trigger a one-off job and confirm startup logs show `perlmutter`, the correct archive root and `dry_run=false`. 2. Check the [scanner results](setup-ingestion-operations.md#check-scanner-results) - and verify links in SimBoard. Handled API failures can occur even when the pod succeeds. + and verify links in SimBoard. Deferred state lookups or failed link submissions + cause a nonzero scanner exit and a failed job. 3. Confirm the next scheduled run and monitor duration: every scan walks the full archive. ### Workload 5: Frontend Deployment (`frontend`) diff --git a/docs/operations/setup-ingestion-operations.md b/docs/operations/setup-ingestion-operations.md index e9e33ce9..77f8cd94 100644 --- a/docs/operations/setup-ingestion-operations.md +++ b/docs/operations/setup-ingestion-operations.md @@ -264,14 +264,22 @@ and verify the machine's archive in `backend/app/scripts/ingestion/diagnostics_a protected API files contain authorized service-account credentials. 2. Merge the template's dev/prod diagnostics entries into the installed crontab. They run daily at 14:00 UTC with `DRY_RUN=false` and separate locks. -3. To run a scan manually, select the API environment: +3. To run a scan manually, run Make from the deployed repository root + (`${SIMBOARD_ROOT}/repository/simboard`) and select the API environment explicitly: ```bash - DRY_RUN=false SIMBOARD_ENV_FILE="${SIMBOARD_ROOT}/operations/env.prod.sh" \ - "${SIMBOARD_ROOT}/repository/simboard/backend/app/scripts/ingestion/sites/site_ingestion_launcher.sh" chrysalis diagnostics + make diagnostics-dry-run SIMBOARD_ROOT="${SIMBOARD_ROOT}" site=chrysalis env=prod + make diagnostics-apply SIMBOARD_ROOT="${SIMBOARD_ROOT}" site=chrysalis env=prod ``` - Use `env.dev.sh` for development or `DRY_RUN=true` for an offline dry scan. + Use `env=dev` for development. The dry-run target scans offline without loading + API credentials; `env` selects log and lock naming. The apply target loads the + protected `operations/env..sh` file and enables live linking. Both targets + use the existing launcher for locks, logs, and exit status. These targets only + link already-published output; use `v3-diagnostics-*` for historical backfill. + Dry-run and apply share the selected environment's diagnostics lock with + scheduled scans. A concurrent run fails without starting the scanner; check + the indicated log location for lock contention or scanner failures. 4. Review `operations/raw_logs/simboard-diagnostics-chrysalis-*.log` and verify links in SimBoard. Each scan walks the full archive; ingestion year/case limits do not apply. @@ -287,9 +295,9 @@ Configure a separate CronJob through Rancher using the - `diagnostics_scanner_completed` reports the outcome and counters collected so far. - Fatal errors log `diagnostics_scanner_failed` and exit nonzero. Discovery failures may leave the candidate count at zero; abrupt termination may omit the summary. -- Check `deferred_state_lookups` and `failed_link_submissions` even when the run - succeeds: handled API failures are counted without a nonzero exit. Later scans - retry deferred work and skip unchanged links. +- Nonzero `deferred_state_lookups` or `failed_link_submissions` also cause a + nonzero exit and a failed summary outcome. Later scans retry deferred work and + skip unchanged links. ## Operate safely diff --git a/docs/operations/test-ingestion-operations.md b/docs/operations/test-ingestion-operations.md index 70968600..cf1b4373 100644 --- a/docs/operations/test-ingestion-operations.md +++ b/docs/operations/test-ingestion-operations.md @@ -104,16 +104,19 @@ Confirm two `chrysalis diagnostics` entries at `0 14 * * *`, using `env.dev.sh` and `env.prod.sh` with `DRY_RUN=false`. Do not execute these live jobs with dummy credentials or without the site archive. Staging/archive schedules are unchanged. -Run the isolated launcher/scanner tests from the repository root: +Run the isolated Make wrapper/launcher/scanner tests from the repository root: ```bash uv run --project backend pytest \ + backend/tests/features/ingestion/test_diagnostics_make_targets.py \ backend/tests/features/ingestion/test_site_collection_launcher.py \ backend/tests/features/ingestion/test_diagnostics_link_scanner.py \ --noconftest --no-cov ``` These tests use temporary archives and mocked runtimes/API calls. +The Make wrapper tests verify offline dry runs, dev/prod selection, argument +validation, and failure propagation without running real scans. `--noconftest` skips the unused application-wide database setup. ## 5. Test a no-op refresh