diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 0e65453..148fa21 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -90,3 +90,7 @@ jobs: - name: E2E test (mocked API, no backend needed) working-directory: apps/web run: npx playwright test + + - name: Build and test the static demo export + working-directory: apps/web + run: npm run build:demo && npm run test:e2e:demo diff --git a/apps/web/app/evaluations/[id]/evaluation-view.tsx b/apps/web/app/evaluations/[id]/evaluation-view.tsx new file mode 100644 index 0000000..bb4a677 --- /dev/null +++ b/apps/web/app/evaluations/[id]/evaluation-view.tsx @@ -0,0 +1,169 @@ +"use client"; + +import Link from "next/link"; +import { useEffect, useMemo, useState } from "react"; +import { getEvaluation, type CaseStatus, type EvaluationRunDetailView } from "@/lib/api-client"; + +const METRIC_LABELS: Record = { + root_cause_accuracy: "Root cause accuracy", + dimension_accuracy: "Dimension accuracy", + evidence_citation_validity_rate: "Evidence citation validity", + tool_execution_success_rate: "Tool execution success", + unsupported_claim_rate: "Unsupported claim rate", + abstention_accuracy: "Abstention accuracy", + average_tool_calls: "Avg. tool calls", + average_latency_ms: "Avg. latency (ms)", + completion_rate: "Completion rate", + scenario_count: "Scenarios", +}; + +const STATUS_FILTERS: (CaseStatus | "all")[] = [ + "all", + "pass", + "fail", + "abstained", + "execution_error", +]; + +const STATUS_STYLES: Record = { + pass: "text-positive", + fail: "text-negative", + abstained: "text-caution", + execution_error: "text-faint", +}; + +function formatMetricValue(key: string, value: number): string { + if (key.endsWith("_rate") || key.endsWith("_accuracy")) { + return `${(value * 100).toFixed(1)}%`; + } + if (key === "average_latency_ms") { + return value.toFixed(0); + } + if (key === "average_tool_calls") { + return value.toFixed(1); + } + return String(value); +} + +export function EvaluationView({ id }: { id: string }) { + const [run, setRun] = useState(null); + const [error, setError] = useState(null); + const [filter, setFilter] = useState("all"); + + useEffect(() => { + getEvaluation(id) + .then(setRun) + .catch((err: unknown) => + setError(err instanceof Error ? err.message : "Failed to load evaluation run"), + ); + }, [id]); + + const filteredCases = useMemo(() => { + if (!run) return []; + return filter === "all" ? run.cases : run.cases.filter((c) => c.status === filter); + }, [run, filter]); + + if (error) { + return ( +
+

+ {error} +

+
+ ); + } + + if (!run) { + return ( +
+

Loading…

+
+ ); + } + + return ( +
+
+

+ {run.model_name} — {run.split} +

+

+ {new Date(run.started_at).toLocaleString()} · {run.status} +

+
+ + {run.aggregate_metrics && ( +
+ {Object.entries(run.aggregate_metrics).map(([key, value]) => ( +
+

{METRIC_LABELS[key] ?? key}

+

+ {formatMetricValue(key, value)} +

+
+ ))} +
+ )} + +
+ {STATUS_FILTERS.map((status) => ( + + ))} +
+ +
+ + + + {["Scenario", "Template", "Predicted driver", "Status", "Replay"].map((heading) => ( + + ))} + + + + {filteredCases.map((c) => ( + + + + + + + + ))} + +
+ {heading} +
{c.scenario_id}{c.template ?? "—"} + {c.predicted_primary_driver ?? "—"} + + {c.status} + + {c.investigation_id && !process.env.NEXT_PUBLIC_DEMO_MODE ? ( + + view + + ) : ( + — + )} +
+
+
+ ); +} diff --git a/apps/web/app/evaluations/[id]/page.tsx b/apps/web/app/evaluations/[id]/page.tsx index 059001d..a1ce79a 100644 --- a/apps/web/app/evaluations/[id]/page.tsx +++ b/apps/web/app/evaluations/[id]/page.tsx @@ -1,171 +1,19 @@ -"use client"; - -import Link from "next/link"; -import { use, useEffect, useMemo, useState } from "react"; -import { getEvaluation, type CaseStatus, type EvaluationRunDetailView } from "@/lib/api-client"; - -const METRIC_LABELS: Record = { - root_cause_accuracy: "Root cause accuracy", - dimension_accuracy: "Dimension accuracy", - evidence_citation_validity_rate: "Evidence citation validity", - tool_execution_success_rate: "Tool execution success", - unsupported_claim_rate: "Unsupported claim rate", - abstention_accuracy: "Abstention accuracy", - average_tool_calls: "Avg. tool calls", - average_latency_ms: "Avg. latency (ms)", - completion_rate: "Completion rate", - scenario_count: "Scenarios", -}; - -const STATUS_FILTERS: (CaseStatus | "all")[] = [ - "all", - "pass", - "fail", - "abstained", - "execution_error", -]; - -const STATUS_STYLES: Record = { - pass: "text-positive", - fail: "text-negative", - abstained: "text-caution", - execution_error: "text-faint", -}; - -function formatMetricValue(key: string, value: number): string { - if (key.endsWith("_rate") || key.endsWith("_accuracy")) { - return `${(value * 100).toFixed(1)}%`; - } - if (key === "average_latency_ms") { - return value.toFixed(0); - } - if (key === "average_tool_calls") { - return value.toFixed(1); - } - return String(value); +import fs from "node:fs"; +import path from "node:path"; +import { EvaluationView } from "./evaluation-view"; + +export function generateStaticParams() { + if (!process.env.NEXT_PUBLIC_DEMO_MODE) return []; + const idsPath = path.join(process.cwd(), "public/demo/evaluation-ids.json"); + const ids = JSON.parse(fs.readFileSync(idsPath, "utf-8")) as string[]; + return ids.map((id) => ({ id })); } -export default function EvaluationDetailPage({ params }: { params: Promise<{ id: string }> }) { - const { id } = use(params); - - const [run, setRun] = useState(null); - const [error, setError] = useState(null); - const [filter, setFilter] = useState("all"); - - useEffect(() => { - getEvaluation(id) - .then(setRun) - .catch((err: unknown) => - setError(err instanceof Error ? err.message : "Failed to load evaluation run"), - ); - }, [id]); - - const filteredCases = useMemo(() => { - if (!run) return []; - return filter === "all" ? run.cases : run.cases.filter((c) => c.status === filter); - }, [run, filter]); - - if (error) { - return ( -
-

- {error} -

-
- ); - } - - if (!run) { - return ( -
-

Loading…

-
- ); - } - - return ( -
-
-

- {run.model_name} — {run.split} -

-

- {new Date(run.started_at).toLocaleString()} · {run.status} -

-
- - {run.aggregate_metrics && ( -
- {Object.entries(run.aggregate_metrics).map(([key, value]) => ( -
-

{METRIC_LABELS[key] ?? key}

-

- {formatMetricValue(key, value)} -

-
- ))} -
- )} - -
- {STATUS_FILTERS.map((status) => ( - - ))} -
- -
- - - - {["Scenario", "Template", "Predicted driver", "Status", "Replay"].map((heading) => ( - - ))} - - - - {filteredCases.map((c) => ( - - - - - - - - ))} - -
- {heading} -
{c.scenario_id}{c.template ?? "—"} - {c.predicted_primary_driver ?? "—"} - - {c.status} - - {c.investigation_id ? ( - - view - - ) : ( - — - )} -
-
-
- ); +export default async function EvaluationDetailPage({ + params, +}: { + params: Promise<{ id: string }>; +}) { + const { id } = await params; + return ; } diff --git a/apps/web/app/investigations/[id]/investigation-view.tsx b/apps/web/app/investigations/[id]/investigation-view.tsx new file mode 100644 index 0000000..1cea46d --- /dev/null +++ b/apps/web/app/investigations/[id]/investigation-view.tsx @@ -0,0 +1,346 @@ +"use client"; + +import { useCallback, useEffect, useMemo, useRef, useState } from "react"; +import { EvidenceDrawer } from "@/components/evidence-drawer"; +import { EvidenceLedger, type Exhibit } from "@/components/evidence-ledger"; +import { HypothesisPanel } from "@/components/hypothesis-panel"; +import { + cancelInvestigation, + getEvidence, + getInvestigation, + getInvestigationEvents, + type EvidenceView, + type HypothesisPayload, + type InvestigationEventView, + type InvestigationView, +} from "@/lib/api-client"; +import { formatCurrency, formatPercentChange } from "@/lib/format"; + +const RUNNING_POLL_INTERVAL_MS = 2000; + +const STATUS_STYLES: Record = { + running: "text-signal", + completed: "text-positive", + partial: "text-caution", + failed: "text-negative", + timed_out: "text-negative", + cancelled: "text-faint", +}; + +function latestHypotheses(events: InvestigationEventView[]): HypothesisPayload[] { + const byId = new Map(); + for (const event of events) { + if (event.event_type !== "hypothesis_update") continue; + const hypothesis = event.payload as unknown as HypothesisPayload; + byId.set(hypothesis.id, hypothesis); + } + return Array.from(byId.values()); +} + +/** Every executed query becomes a numbered exhibit, in execution order. + The ordinal — not the uuid — is how a reader refers to evidence, so the + ledger deliberately never renders the raw id. */ +function exhibitsFrom(events: InvestigationEventView[]): Exhibit[] { + const exhibits: Exhibit[] = []; + for (const event of events) { + if (event.event_type !== "tool_call") continue; + const { evidence_id, tool_name, row_count, execution_ms } = event.payload; + if (typeof evidence_id !== "string" || typeof tool_name !== "string") continue; + exhibits.push({ + evidenceId: evidence_id, + ordinal: exhibits.length + 1, + toolName: tool_name, + rowCount: typeof row_count === "number" ? row_count : 0, + executionMs: typeof execution_ms === "number" ? execution_ms : 0, + }); + } + return exhibits; +} + +export function InvestigationView({ id }: { id: string }) { + const [investigation, setInvestigation] = useState(null); + const [events, setEvents] = useState([]); + const [error, setError] = useState(null); + const [isLoading, setIsLoading] = useState(true); + + const [selectedEvidenceId, setSelectedEvidenceId] = useState(null); + const [evidence, setEvidence] = useState(null); + const [isEvidenceLoading, setIsEvidenceLoading] = useState(false); + const [evidenceError, setEvidenceError] = useState(null); + + const [linkedEvidenceId, setLinkedEvidenceId] = useState(null); + + const pollRef = useRef | null>(null); + + const load = useCallback(async () => { + try { + const [investigationData, eventsData] = await Promise.all([ + getInvestigation(id), + getInvestigationEvents(id), + ]); + setInvestigation(investigationData); + setEvents(eventsData); + setError(null); + } catch (err) { + setError(err instanceof Error ? err.message : "Failed to load investigation"); + } finally { + setIsLoading(false); + } + }, [id]); + + useEffect(() => { + load(); + }, [load]); + + useEffect(() => { + if (investigation?.status === "running") { + pollRef.current = setInterval(load, RUNNING_POLL_INTERVAL_MS); + } + return () => { + if (pollRef.current) clearInterval(pollRef.current); + }; + }, [investigation?.status, load]); + + const openEvidence = useCallback( + async (evidenceId: string) => { + setSelectedEvidenceId(evidenceId); + setIsEvidenceLoading(true); + setEvidenceError(null); + try { + const data = await getEvidence(id, evidenceId); + setEvidence(data); + } catch (err) { + setEvidenceError(err instanceof Error ? err.message : "Failed to load evidence"); + } finally { + setIsEvidenceLoading(false); + } + }, + [id], + ); + + const closeEvidence = useCallback(() => { + setSelectedEvidenceId(null); + setEvidence(null); + setEvidenceError(null); + }, []); + + const [isCancelling, setIsCancelling] = useState(false); + const handleCancel = useCallback(async () => { + setIsCancelling(true); + try { + const updated = await cancelInvestigation(id); + setInvestigation(updated); + } catch (err) { + setError(err instanceof Error ? err.message : "Failed to cancel investigation"); + } finally { + setIsCancelling(false); + } + }, [id]); + + const exhibits = useMemo(() => exhibitsFrom(events), [events]); + const ordinalByEvidenceId = useMemo( + () => new Map(exhibits.map((exhibit) => [exhibit.evidenceId, exhibit.ordinal])), + [exhibits], + ); + + if (isLoading) { + return ( +
+

Loading…

+
+ ); + } + + if (error || !investigation) { + return ( +
+

+ {error ?? "Investigation not found"} +

+
+ ); + } + + const hypotheses = latestHypotheses(events); + const report = investigation.report; + + return ( +
+
+
+

{investigation.metric}

+

+ {investigation.current_period.start} → {investigation.current_period.end} +

+

+ compared with {investigation.comparison_period.start} →{" "} + {investigation.comparison_period.end} +

+ {investigation.question && ( +

+ “{investigation.question}” +

+ )} +
+
+ {investigation.status === "running" && ( + + )} + + {investigation.status} + +
+
+ + {/* The two columns are the product's central claim: everything on the + left is an assertion, everything on the right is the executed + query behind it. */} +
+
+
+

Hypotheses

+
+ +
+
+ + {report && ( +
+

Report

+
+

+ {report.headline} +

+ +
+
+

Current

+

+ {formatCurrency(report.observed_change.current_value)} +

+
+
+

Change

+

+ {formatPercentChange(report.observed_change.percent_change)} +

+
+
+ +
    + {report.findings.map((finding, index) => ( + // Findings have no stable id — index is fine, this list + // never reorders after the report is generated. + +
  • +

    {finding.claim}

    +
    + + {finding.claim_type} · {finding.confidence} + + {finding.evidence_ids.map((evidenceId) => { + const ordinal = ordinalByEvidenceId.get(evidenceId); + const isLinked = evidenceId === linkedEvidenceId; + return ( + + ); + })} +
    +
  • + ))} +
+ + {report.limitations.length > 0 && ( +
+

Limitations

+
    + {report.limitations.map((limitation) => ( +
  • + {limitation} +
  • + ))} +
+
+ )} +
+
+ )} +
+ + {/* The right rail is everything the machine did — the step trace and + the queries it ran. The left column is what it is willing to + claim off the back of that. */} + +
+ + {selectedEvidenceId && ( + + )} +
+ ); +} diff --git a/apps/web/app/investigations/[id]/page.tsx b/apps/web/app/investigations/[id]/page.tsx index cc6c2f8..0c57a9d 100644 --- a/apps/web/app/investigations/[id]/page.tsx +++ b/apps/web/app/investigations/[id]/page.tsx @@ -1,348 +1,19 @@ -"use client"; - -import { use, useCallback, useEffect, useMemo, useRef, useState } from "react"; -import { EvidenceDrawer } from "@/components/evidence-drawer"; -import { EvidenceLedger, type Exhibit } from "@/components/evidence-ledger"; -import { HypothesisPanel } from "@/components/hypothesis-panel"; -import { - cancelInvestigation, - getEvidence, - getInvestigation, - getInvestigationEvents, - type EvidenceView, - type HypothesisPayload, - type InvestigationEventView, - type InvestigationView, -} from "@/lib/api-client"; -import { formatCurrency, formatPercentChange } from "@/lib/format"; - -const RUNNING_POLL_INTERVAL_MS = 2000; - -const STATUS_STYLES: Record = { - running: "text-signal", - completed: "text-positive", - partial: "text-caution", - failed: "text-negative", - timed_out: "text-negative", - cancelled: "text-faint", -}; - -function latestHypotheses(events: InvestigationEventView[]): HypothesisPayload[] { - const byId = new Map(); - for (const event of events) { - if (event.event_type !== "hypothesis_update") continue; - const hypothesis = event.payload as unknown as HypothesisPayload; - byId.set(hypothesis.id, hypothesis); - } - return Array.from(byId.values()); +import fs from "node:fs"; +import path from "node:path"; +import { InvestigationView } from "./investigation-view"; + +export function generateStaticParams() { + if (!process.env.NEXT_PUBLIC_DEMO_MODE) return []; + const idsPath = path.join(process.cwd(), "public/demo/investigation-ids.json"); + const ids = JSON.parse(fs.readFileSync(idsPath, "utf-8")) as string[]; + return ids.map((id) => ({ id })); } -/** Every executed query becomes a numbered exhibit, in execution order. - The ordinal — not the uuid — is how a reader refers to evidence, so the - ledger deliberately never renders the raw id. */ -function exhibitsFrom(events: InvestigationEventView[]): Exhibit[] { - const exhibits: Exhibit[] = []; - for (const event of events) { - if (event.event_type !== "tool_call") continue; - const { evidence_id, tool_name, row_count, execution_ms } = event.payload; - if (typeof evidence_id !== "string" || typeof tool_name !== "string") continue; - exhibits.push({ - evidenceId: evidence_id, - ordinal: exhibits.length + 1, - toolName: tool_name, - rowCount: typeof row_count === "number" ? row_count : 0, - executionMs: typeof execution_ms === "number" ? execution_ms : 0, - }); - } - return exhibits; -} - -export default function InvestigationPage({ params }: { params: Promise<{ id: string }> }) { - const { id } = use(params); - - const [investigation, setInvestigation] = useState(null); - const [events, setEvents] = useState([]); - const [error, setError] = useState(null); - const [isLoading, setIsLoading] = useState(true); - - const [selectedEvidenceId, setSelectedEvidenceId] = useState(null); - const [evidence, setEvidence] = useState(null); - const [isEvidenceLoading, setIsEvidenceLoading] = useState(false); - const [evidenceError, setEvidenceError] = useState(null); - - const [linkedEvidenceId, setLinkedEvidenceId] = useState(null); - - const pollRef = useRef | null>(null); - - const load = useCallback(async () => { - try { - const [investigationData, eventsData] = await Promise.all([ - getInvestigation(id), - getInvestigationEvents(id), - ]); - setInvestigation(investigationData); - setEvents(eventsData); - setError(null); - } catch (err) { - setError(err instanceof Error ? err.message : "Failed to load investigation"); - } finally { - setIsLoading(false); - } - }, [id]); - - useEffect(() => { - load(); - }, [load]); - - useEffect(() => { - if (investigation?.status === "running") { - pollRef.current = setInterval(load, RUNNING_POLL_INTERVAL_MS); - } - return () => { - if (pollRef.current) clearInterval(pollRef.current); - }; - }, [investigation?.status, load]); - - const openEvidence = useCallback( - async (evidenceId: string) => { - setSelectedEvidenceId(evidenceId); - setIsEvidenceLoading(true); - setEvidenceError(null); - try { - const data = await getEvidence(id, evidenceId); - setEvidence(data); - } catch (err) { - setEvidenceError(err instanceof Error ? err.message : "Failed to load evidence"); - } finally { - setIsEvidenceLoading(false); - } - }, - [id], - ); - - const closeEvidence = useCallback(() => { - setSelectedEvidenceId(null); - setEvidence(null); - setEvidenceError(null); - }, []); - - const [isCancelling, setIsCancelling] = useState(false); - const handleCancel = useCallback(async () => { - setIsCancelling(true); - try { - const updated = await cancelInvestigation(id); - setInvestigation(updated); - } catch (err) { - setError(err instanceof Error ? err.message : "Failed to cancel investigation"); - } finally { - setIsCancelling(false); - } - }, [id]); - - const exhibits = useMemo(() => exhibitsFrom(events), [events]); - const ordinalByEvidenceId = useMemo( - () => new Map(exhibits.map((exhibit) => [exhibit.evidenceId, exhibit.ordinal])), - [exhibits], - ); - - if (isLoading) { - return ( -
-

Loading…

-
- ); - } - - if (error || !investigation) { - return ( -
-

- {error ?? "Investigation not found"} -

-
- ); - } - - const hypotheses = latestHypotheses(events); - const report = investigation.report; - - return ( -
-
-
-

{investigation.metric}

-

- {investigation.current_period.start} → {investigation.current_period.end} -

-

- compared with {investigation.comparison_period.start} →{" "} - {investigation.comparison_period.end} -

- {investigation.question && ( -

- “{investigation.question}” -

- )} -
-
- {investigation.status === "running" && ( - - )} - - {investigation.status} - -
-
- - {/* The two columns are the product's central claim: everything on the - left is an assertion, everything on the right is the executed - query behind it. */} -
-
-
-

Hypotheses

-
- -
-
- - {report && ( -
-

Report

-
-

- {report.headline} -

- -
-
-

Current

-

- {formatCurrency(report.observed_change.current_value)} -

-
-
-

Change

-

- {formatPercentChange(report.observed_change.percent_change)} -

-
-
- -
    - {report.findings.map((finding, index) => ( - // Findings have no stable id — index is fine, this list - // never reorders after the report is generated. - -
  • -

    {finding.claim}

    -
    - - {finding.claim_type} · {finding.confidence} - - {finding.evidence_ids.map((evidenceId) => { - const ordinal = ordinalByEvidenceId.get(evidenceId); - const isLinked = evidenceId === linkedEvidenceId; - return ( - - ); - })} -
    -
  • - ))} -
- - {report.limitations.length > 0 && ( -
-

Limitations

-
    - {report.limitations.map((limitation) => ( -
  • - {limitation} -
  • - ))} -
-
- )} -
-
- )} -
- - {/* The right rail is everything the machine did — the step trace and - the queries it ran. The left column is what it is willing to - claim off the back of that. */} - -
- - {selectedEvidenceId && ( - - )} -
- ); +export default async function InvestigationPage({ + params, +}: { + params: Promise<{ id: string }>; +}) { + const { id } = await params; + return ; } diff --git a/apps/web/app/layout.tsx b/apps/web/app/layout.tsx index 66659ef..5cbab23 100644 --- a/apps/web/app/layout.tsx +++ b/apps/web/app/layout.tsx @@ -1,6 +1,7 @@ import type { Metadata } from "next"; import { IBM_Plex_Mono, IBM_Plex_Sans, IBM_Plex_Serif } from "next/font/google"; import Link from "next/link"; +import { DemoBanner } from "@/components/demo-banner"; import "./globals.css"; // IBM Plex across three roles, matching the three registers the product @@ -67,6 +68,7 @@ export default function RootLayout({ + {process.env.NEXT_PUBLIC_DEMO_MODE && }
diff --git a/apps/web/app/page.tsx b/apps/web/app/page.tsx index 05a39d7..4e3849d 100644 --- a/apps/web/app/page.tsx +++ b/apps/web/app/page.tsx @@ -45,10 +45,12 @@ function DateField({ label, value, onChange, + disabled, }: { label: string; value: string; onChange: (value: string) => void; + disabled?: boolean; }) { return ( ); @@ -65,6 +68,7 @@ function DateField({ export default function Home() { const router = useRouter(); + const isDemoMode = Boolean(process.env.NEXT_PUBLIC_DEMO_MODE); const [currentStart, setCurrentStart] = useState(DEFAULT_CURRENT_START); const [currentEnd, setCurrentEnd] = useState(DEFAULT_CURRENT_END); @@ -144,13 +148,33 @@ export default function Home() {
This period - - + +
Compared with - - + +
@@ -190,7 +214,7 @@ export default function Home() {
diff --git a/apps/web/components/demo-banner.tsx b/apps/web/components/demo-banner.tsx new file mode 100644 index 0000000..f37db65 --- /dev/null +++ b/apps/web/components/demo-banner.tsx @@ -0,0 +1,17 @@ +import fs from "node:fs"; +import path from "node:path"; + +export function DemoBanner() { + const metaPath = path.join(process.cwd(), "public/demo/captured-at.json"); + const { captured_at: capturedAt } = JSON.parse(fs.readFileSync(metaPath, "utf-8")) as { + captured_at: string; + }; + + return ( +
+ Read-only demo — investigations captured {capturedAt}. The investigation + loop needs a local LLM, so this replays real runs instead of computing + new ones. +
+ ); +} diff --git a/apps/web/e2e/demo.spec.ts b/apps/web/e2e/demo.spec.ts new file mode 100644 index 0000000..20f8f0e --- /dev/null +++ b/apps/web/e2e/demo.spec.ts @@ -0,0 +1,27 @@ +import { expect, test } from "@playwright/test"; + +test("landing investigation renders with a working citation, and controls are disabled", async ({ + page, +}) => { + await page.goto("/"); + + await expect(page.getByText(/Read-only demo/)).toBeVisible(); + await expect( + page.getByRole("button", { name: "Investigate revenue change" }), + ).toBeDisabled(); + + await page.locator('a[href^="/investigations/"]').first().click(); + + await expect(page.getByTestId("investigation-status")).toHaveText("completed"); + + // Two elements match "E1": the finding's inline citation and the sidebar + // Evidence Ledger's exhibit entry — both call openEvidence() for the same + // evidence id, so .first() (the inline citation, first in DOM order) + // exercises the same citation flow deterministically. + const firstCitation = page.getByRole("button", { name: /^E1/ }).first(); + await firstCitation.click(); + + const dialog = page.getByRole("dialog", { name: "Evidence detail" }); + await expect(dialog).toBeVisible(); + await expect(dialog.getByText("rows")).toBeVisible(); +}); diff --git a/apps/web/lib/__tests__/api-client.test.ts b/apps/web/lib/__tests__/api-client.test.ts index 9c84b5b..7ed9cf8 100644 --- a/apps/web/lib/__tests__/api-client.test.ts +++ b/apps/web/lib/__tests__/api-client.test.ts @@ -6,6 +6,7 @@ import { getEvidence, getInvestigation, getInvestigationEvents, + getInvestigations, getMetricsSummary, } from "@/lib/api-client"; @@ -215,3 +216,28 @@ describe("cancelInvestigation", () => { expect(init).toEqual({ method: "POST" }); }); }); + +describe("fetchJson in demo mode", () => { + afterEach(() => { + vi.unstubAllGlobals(); + delete process.env.NEXT_PUBLIC_DEMO_MODE; + }); + + it("resolves from the demo fixture manifest instead of calling the real API", async () => { + process.env.NEXT_PUBLIC_DEMO_MODE = "1"; + const manifest = [{ key: "GET /api/investigations", file: "investigations-list.json" }]; + const fixture = [{ investigation_id: "inv-1", status: "completed" }]; + + const fetchMock = vi.fn((url: string) => { + if (url === "/demo/manifest.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(manifest) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve(fixture) }); + }); + vi.stubGlobal("fetch", fetchMock); + + const result = await getInvestigations(); + expect(result).toEqual(fixture); + expect(fetchMock.mock.calls[0][0]).toBe("/demo/manifest.json"); + }); +}); diff --git a/apps/web/lib/__tests__/demo-data.test.ts b/apps/web/lib/__tests__/demo-data.test.ts new file mode 100644 index 0000000..7e79c27 --- /dev/null +++ b/apps/web/lib/__tests__/demo-data.test.ts @@ -0,0 +1,69 @@ +import { afterEach, describe, expect, it, vi } from "vitest"; +import { buildDemoKey, resolveDemoFixture } from "@/lib/demo-data"; + +describe("buildDemoKey", () => { + it("returns a plain method-and-path key when there are no params", () => { + expect(buildDemoKey("GET", "/api/investigations")).toBe("GET /api/investigations"); + }); + + it("sorts params so callers passing them in a different order collide on the same key", () => { + const a = buildDemoKey("GET", "/api/metrics/summary", { + current_start: "2018-01-01", + comparison_start: "2017-12-01", + }); + const b = buildDemoKey("GET", "/api/metrics/summary", { + comparison_start: "2017-12-01", + current_start: "2018-01-01", + }); + expect(a).toBe(b); + expect(a).toBe( + "GET /api/metrics/summary?comparison_start=2017-12-01¤t_start=2018-01-01", + ); + }); +}); + +describe("resolveDemoFixture", () => { + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it("fetches the manifest, then the matching fixture file", async () => { + const manifest = [{ key: "GET /api/investigations", file: "investigations-list.json" }]; + const fixture = [{ investigation_id: "inv-1", status: "completed" }]; + + vi.stubGlobal( + "fetch", + vi.fn((url: string) => { + if (url === "/demo/manifest.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(manifest) }); + } + if (url === "/demo/investigations-list.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(fixture) }); + } + throw new Error(`unexpected fetch: ${url}`); + }), + ); + + const result = await resolveDemoFixture("GET", "/api/investigations"); + expect(result).toEqual(fixture); + }); + + it("throws when no manifest entry matches the request", async () => { + vi.stubGlobal( + "fetch", + vi.fn().mockResolvedValue({ ok: true, json: () => Promise.resolve([]) }), + ); + + await expect(resolveDemoFixture("GET", "/api/investigations")).rejects.toThrow( + "Demo fixture not found for GET /api/investigations", + ); + }); + + it("throws when the manifest itself fails to load", async () => { + vi.stubGlobal("fetch", vi.fn().mockResolvedValue({ ok: false, status: 500 })); + + await expect(resolveDemoFixture("GET", "/api/investigations")).rejects.toThrow( + "Failed to load /demo/manifest.json: 500", + ); + }); +}); diff --git a/apps/web/lib/api-client.ts b/apps/web/lib/api-client.ts index 23a4d55..81fd027 100644 --- a/apps/web/lib/api-client.ts +++ b/apps/web/lib/api-client.ts @@ -1,3 +1,5 @@ +import { resolveDemoFixture } from "@/lib/demo-data"; + const API_BASE_URL = process.env.NEXT_PUBLIC_API_BASE_URL ?? "http://localhost:8000"; export interface MetricComparison { @@ -147,6 +149,10 @@ async function fetchJson( params?: Record, init?: RequestInit, ): Promise { + if (process.env.NEXT_PUBLIC_DEMO_MODE) { + return resolveDemoFixture(init?.method ?? "GET", path, params); + } + const url = new URL(path, API_BASE_URL); if (params) { for (const [key, value] of Object.entries(params)) { diff --git a/apps/web/lib/demo-data.ts b/apps/web/lib/demo-data.ts new file mode 100644 index 0000000..da41713 --- /dev/null +++ b/apps/web/lib/demo-data.ts @@ -0,0 +1,41 @@ +interface DemoManifestEntry { + key: string; + file: string; +} + +export function buildDemoKey( + method: string, + path: string, + params?: Record, +): string { + if (!params || Object.keys(params).length === 0) { + return `${method} ${path}`; + } + const query = Object.keys(params) + .sort() + .map((key) => `${key}=${params[key]}`) + .join("&"); + return `${method} ${path}?${query}`; +} + +async function fetchDemoJson(url: string): Promise { + const response = await fetch(url); + if (!response.ok) { + throw new Error(`Failed to load ${url}: ${response.status}`); + } + return response.json() as Promise; +} + +export async function resolveDemoFixture( + method: string, + path: string, + params?: Record, +): Promise { + const key = buildDemoKey(method, path, params); + const manifest = await fetchDemoJson("/demo/manifest.json"); + const entry = manifest.find((item) => item.key === key); + if (!entry) { + throw new Error(`Demo fixture not found for ${key}`); + } + return fetchDemoJson(`/demo/${entry.file}`); +} diff --git a/apps/web/next.config.ts b/apps/web/next.config.ts index e9ffa30..2cde8cc 100644 --- a/apps/web/next.config.ts +++ b/apps/web/next.config.ts @@ -1,7 +1,7 @@ import type { NextConfig } from "next"; const nextConfig: NextConfig = { - /* config options here */ + ...(process.env.NEXT_PUBLIC_DEMO_MODE ? { output: "export" } : {}), }; export default nextConfig; diff --git a/apps/web/package.json b/apps/web/package.json index b33cbaf..8afb7d1 100644 --- a/apps/web/package.json +++ b/apps/web/package.json @@ -5,11 +5,13 @@ "scripts": { "dev": "next dev --turbopack", "build": "next build --turbopack", + "build:demo": "NEXT_PUBLIC_DEMO_MODE=1 next build --turbopack", "start": "next start", "lint": "eslint", "test": "vitest run", "test:coverage": "vitest run --coverage", "test:e2e": "playwright test", + "test:e2e:demo": "playwright test --config=playwright.demo.config.ts", "format": "prettier --write .", "typecheck": "tsc --noEmit" }, diff --git a/apps/web/playwright.config.ts b/apps/web/playwright.config.ts index 8044b20..62beed6 100644 --- a/apps/web/playwright.config.ts +++ b/apps/web/playwright.config.ts @@ -2,6 +2,7 @@ import { defineConfig, devices } from "@playwright/test"; export default defineConfig({ testDir: "./e2e", + testIgnore: "demo.spec.ts", fullyParallel: true, forbidOnly: !!process.env.CI, retries: process.env.CI ? 2 : 0, diff --git a/apps/web/playwright.demo.config.ts b/apps/web/playwright.demo.config.ts new file mode 100644 index 0000000..f21abc1 --- /dev/null +++ b/apps/web/playwright.demo.config.ts @@ -0,0 +1,21 @@ +import { defineConfig, devices } from "@playwright/test"; + +export default defineConfig({ + testDir: "./e2e", + testMatch: "demo.spec.ts", + fullyParallel: true, + forbidOnly: !!process.env.CI, + retries: process.env.CI ? 2 : 0, + reporter: "list", + use: { + baseURL: "http://localhost:4173", + trace: "retain-on-failure", + }, + webServer: { + command: "npx --yes serve@14.2.6 out -l 4173", + url: "http://localhost:4173", + reuseExistingServer: !process.env.CI, + timeout: 60_000, + }, + projects: [{ name: "chromium", use: { ...devices["Desktop Chrome"] } }], +}); diff --git a/apps/web/public/demo/captured-at.json b/apps/web/public/demo/captured-at.json new file mode 100644 index 0000000..4b4a8e4 --- /dev/null +++ b/apps/web/public/demo/captured-at.json @@ -0,0 +1,3 @@ +{ + "captured_at": "2026-09-12" +} \ No newline at end of file diff --git a/apps/web/public/demo/evaluation-f51f692d-f106-430c-879f-dce4019e091a.json b/apps/web/public/demo/evaluation-f51f692d-f106-430c-879f-dce4019e091a.json new file mode 100644 index 0000000..67eff79 --- /dev/null +++ b/apps/web/public/demo/evaluation-f51f692d-f106-430c-879f-dce4019e091a.json @@ -0,0 +1,275 @@ +{ + "run_id": "f51f692d-f106-430c-879f-dce4019e091a", + "model_name": "qwen3:8b", + "split": "held_out", + "status": "completed", + "started_at": "2026-09-13T19:06:31.134563", + "completed_at": "2026-09-13T19:19:16.139806", + "aggregate_metrics": { + "scenario_count": 11, + "completion_rate": 1.0, + "average_latency_ms": 69406.31684071956, + "average_tool_calls": 4.363636363636363, + "dimension_accuracy": 0.0, + "abstention_accuracy": 0.8181818181818182, + "root_cause_accuracy": 0.5555555555555556, + "unsupported_claim_rate": 0.0, + "tool_execution_success_rate": 1.0, + "evidence_citation_validity_rate": 1.0 + }, + "cases": [ + { + "scenario_id": "cs-08", + "template": "cancellation_spike", + "is_unanswerable": false, + "investigation_id": "304f80f1-647e-4e80-bb17-8a3b4fdb26b5", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": true, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 1, + "tool_call_count": 5, + "latency_ms": 59094.81295797741, + "status": "fail", + "notes": null + }, + { + "scenario_id": "cs-09", + "template": "cancellation_spike", + "is_unanswerable": false, + "investigation_id": "1827e3d7-af03-4a9f-a2fe-391ff7d305e6", + "predicted_direction": "increase", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": true, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 2, + "tool_call_count": 4, + "latency_ms": 31792.03549999511, + "status": "fail", + "notes": null + }, + { + "scenario_id": "cs-10", + "template": "cancellation_spike", + "is_unanswerable": false, + "investigation_id": "e73aa416-f4b6-4d6b-a080-e97477ba2b3e", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "product_category": "eletroportateis" + }, + "correct_driver": true, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 0, + "tool_call_count": 4, + "latency_ms": 36406.74308303278, + "status": "fail", + "notes": null + }, + { + "scenario_id": "ov-08", + "template": "order_volume_decline", + "is_unanswerable": false, + "investigation_id": "bde4fb11-518b-4c75-b6a1-430e1d757543", + "predicted_direction": "increase", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": false, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 0, + "tool_call_count": 4, + "latency_ms": 81201.94604096469, + "status": "fail", + "notes": null + }, + { + "scenario_id": "ov-09", + "template": "order_volume_decline", + "is_unanswerable": false, + "investigation_id": "bafc43b1-30be-4c75-a128-e99a945f6495", + "predicted_direction": "increase", + "predicted_primary_driver": "order_volume", + "predicted_dimensions": { + "product_category": "pcs" + }, + "correct_driver": true, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 2, + "tool_call_count": 4, + "latency_ms": 135629.9662500387, + "status": "fail", + "notes": null + }, + { + "scenario_id": "ov-10", + "template": "order_volume_decline", + "is_unanswerable": false, + "investigation_id": "aff55d10-5894-4e3e-90f3-a3e6eb6f30fd", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "product_category": "relogios_presentes" + }, + "correct_driver": false, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 0, + "tool_call_count": 5, + "latency_ms": 106626.45212502684, + "status": "fail", + "notes": null + }, + { + "scenario_id": "sc-08", + "template": "seller_category_decline", + "is_unanswerable": false, + "investigation_id": "3ad51cc2-0ce5-4ff5-b13b-5855ad5ab549", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": false, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 0, + "tool_call_count": 4, + "latency_ms": 59418.19758299971, + "status": "fail", + "notes": null + }, + { + "scenario_id": "sc-09", + "template": "seller_category_decline", + "is_unanswerable": false, + "investigation_id": "b8d6e30e-5baa-4e89-8567-235521c44c5d", + "predicted_direction": "increase", + "predicted_primary_driver": "order_volume", + "predicted_dimensions": { + "product_category": "cama_mesa_banho" + }, + "correct_driver": true, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 1, + "tool_call_count": 5, + "latency_ms": 90786.42745898105, + "status": "fail", + "notes": null + }, + { + "scenario_id": "sc-10", + "template": "seller_category_decline", + "is_unanswerable": false, + "investigation_id": "7c11be0d-c7dc-4308-a597-6f3cb9cf6385", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": false, + "correct_dimension": false, + "abstained": false, + "expected_abstain": false, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 0, + "tool_call_count": 4, + "latency_ms": 37079.35149996774, + "status": "fail", + "notes": null + }, + { + "scenario_id": "ua-04", + "template": "order_volume_decline", + "is_unanswerable": true, + "investigation_id": "cb9a5bd7-07e9-4aab-b263-db6639b6a6bc", + "predicted_direction": "decrease", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "product_category": "relogios_presentes" + }, + "correct_driver": null, + "correct_dimension": null, + "abstained": false, + "expected_abstain": true, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 2, + "tool_call_count": 4, + "latency_ms": 60199.17266594712, + "status": "fail", + "notes": null + }, + { + "scenario_id": "ua-05", + "template": "order_volume_decline", + "is_unanswerable": true, + "investigation_id": "4f06c87f-1195-4f14-b332-6942adac418f", + "predicted_direction": "increase", + "predicted_primary_driver": "cancellation", + "predicted_dimensions": { + "customer_state": "SP" + }, + "correct_driver": null, + "correct_dimension": null, + "abstained": false, + "expected_abstain": true, + "evidence_citation_valid": true, + "tool_execution_success": true, + "unsupported_claim_count": 0, + "total_claim_count": 1, + "tool_call_count": 5, + "latency_ms": 65234.38008298399, + "status": "fail", + "notes": null + } + ] +} \ No newline at end of file diff --git a/apps/web/public/demo/evaluation-ids.json b/apps/web/public/demo/evaluation-ids.json new file mode 100644 index 0000000..3706be2 --- /dev/null +++ b/apps/web/public/demo/evaluation-ids.json @@ -0,0 +1 @@ +["f51f692d-f106-430c-879f-dce4019e091a"] \ No newline at end of file diff --git a/apps/web/public/demo/evaluations-list.json b/apps/web/public/demo/evaluations-list.json new file mode 100644 index 0000000..13d9d6c --- /dev/null +++ b/apps/web/public/demo/evaluations-list.json @@ -0,0 +1,22 @@ +[ + { + "run_id": "f51f692d-f106-430c-879f-dce4019e091a", + "model_name": "qwen3:8b", + "split": "held_out", + "status": "completed", + "started_at": "2026-09-13T19:06:31.134563", + "completed_at": "2026-09-13T19:19:16.139806", + "aggregate_metrics": { + "scenario_count": 11, + "completion_rate": 1.0, + "average_latency_ms": 69406.31684071956, + "average_tool_calls": 4.363636363636363, + "dimension_accuracy": 0.0, + "abstention_accuracy": 0.8181818181818182, + "root_cause_accuracy": 0.5555555555555556, + "unsupported_claim_rate": 0.0, + "tool_execution_success_rate": 1.0, + "evidence_citation_validity_rate": 1.0 + } + } +] \ No newline at end of file diff --git a/apps/web/public/demo/evidence-00d6a3fb-019f-4952-bbfe-bc46bbdc7106.json b/apps/web/public/demo/evidence-00d6a3fb-019f-4952-bbfe-bc46bbdc7106.json new file mode 100644 index 0000000..d2f81f9 --- /dev/null +++ b/apps/web/public/demo/evidence-00d6a3fb-019f-4952-bbfe-bc46bbdc7106.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "00d6a3fb-019f-4952-bbfe-bc46bbdc7106", + "tool_name": "compare_periods", + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 0.015309155766944114, + "percent_change": 0.3255177368212446, + "absolute_change": 0.003759588875702536, + "comparison_value": 0.011549566891241578 + } + ], + "row_count": 1, + "execution_ms": 13.579958991613239, + "warnings": [], + "created_at": "2026-09-12T06:31:56.633801" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-0d5a34b0-4b86-48a9-aa49-887d234cda61.json b/apps/web/public/demo/evidence-0d5a34b0-4b86-48a9-aa49-887d234cda61.json new file mode 100644 index 0000000..7c39b58 --- /dev/null +++ b/apps/web/public/demo/evidence-0d5a34b0-4b86-48a9-aa49-887d234cda61.json @@ -0,0 +1,79 @@ +{ + "evidence_id": "0d5a34b0-4b86-48a9-aa49-887d234cda61", + "tool_name": "calculate_contribution", + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "rows": [ + { + "change": 15359.749999999993, + "segment_value": "PR", + "share_of_total_change": 1.2246729368390235 + }, + { + "change": 12794.570000000007, + "segment_value": "SP", + "share_of_total_change": 1.02014444359397 + }, + { + "change": -8453.86, + "segment_value": "MG", + "share_of_total_change": -0.6740483115822817 + }, + { + "change": 7940.32, + "segment_value": "SC", + "share_of_total_change": 0.6331024276984741 + }, + { + "change": 5432.899999999994, + "segment_value": "RJ", + "share_of_total_change": 0.43317928993328186 + }, + { + "change": -5198.779999999999, + "segment_value": "PE", + "share_of_total_change": -0.4145122915789632 + }, + { + "change": -5058.630000000005, + "segment_value": "RS", + "share_of_total_change": -0.40333776646638114 + }, + { + "change": -4872.84, + "segment_value": "BA", + "share_of_total_change": -0.38852424509166295 + }, + { + "change": 4217.64, + "segment_value": "PI", + "share_of_total_change": 0.3362834398561006 + }, + { + "change": 2905.66, + "segment_value": "MT", + "share_of_total_change": 0.23167585186319298 + } + ], + "row_count": 10, + "execution_ms": 70.6508329603821, + "warnings": [], + "created_at": "2026-09-12T06:45:02.463282" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-3be66378-ed39-4865-92f1-2e8a7c83bca7.json b/apps/web/public/demo/evidence-3be66378-ed39-4865-92f1-2e8a7c83bca7.json new file mode 100644 index 0000000..4c24aa7 --- /dev/null +++ b/apps/web/public/demo/evidence-3be66378-ed39-4865-92f1-2e8a7c83bca7.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "3be66378-ed39-4865-92f1-2e8a7c83bca7", + "tool_name": "compare_periods", + "params": { + "metric": "orders", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 7189.0, + "percent_change": 0.27849902187444425, + "absolute_change": 1566.0, + "comparison_value": 5623.0 + } + ], + "row_count": 1, + "execution_ms": 11.721291986759752, + "warnings": [], + "created_at": "2026-09-12T06:43:05.643544" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-6d536dc5-14a1-44e9-803a-3966949e6d3f.json b/apps/web/public/demo/evidence-6d536dc5-14a1-44e9-803a-3966949e6d3f.json new file mode 100644 index 0000000..a64e0a2 --- /dev/null +++ b/apps/web/public/demo/evidence-6d536dc5-14a1-44e9-803a-3966949e6d3f.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "6d536dc5-14a1-44e9-803a-3966949e6d3f", + "tool_name": "compare_periods", + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 0.011549566891241578, + "percent_change": 0.21420168965886, + "absolute_change": 0.0020375006590767023, + "comparison_value": 0.009512066232164875 + } + ], + "row_count": 1, + "execution_ms": 11.227999988477677, + "warnings": [], + "created_at": "2026-09-12T06:43:05.910937" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-7764dd7b-f478-4bde-8610-64ca1e93dc35.json b/apps/web/public/demo/evidence-7764dd7b-f478-4bde-8610-64ca1e93dc35.json new file mode 100644 index 0000000..4d56b9f --- /dev/null +++ b/apps/web/public/demo/evidence-7764dd7b-f478-4bde-8610-64ca1e93dc35.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "7764dd7b-f478-4bde-8610-64ca1e93dc35", + "tool_name": "compare_periods", + "params": { + "metric": "orders", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 6919.0, + "percent_change": -0.03473772321428571, + "absolute_change": -249.0, + "comparison_value": 7168.0 + } + ], + "row_count": 1, + "execution_ms": 13.631290988996625, + "warnings": [], + "created_at": "2026-09-12T06:44:08.186147" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-8595a0fb-6e94-4f97-b610-67cde1cf001e.json b/apps/web/public/demo/evidence-8595a0fb-6e94-4f97-b610-67cde1cf001e.json new file mode 100644 index 0000000..b483957 --- /dev/null +++ b/apps/web/public/demo/evidence-8595a0fb-6e94-4f97-b610-67cde1cf001e.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "8595a0fb-6e94-4f97-b610-67cde1cf001e", + "tool_name": "compare_periods", + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 0.002882259691598213, + "percent_change": -0.5166517526484949, + "absolute_change": -0.003080852220757909, + "comparison_value": 0.005963111912356122 + } + ], + "row_count": 1, + "execution_ms": 13.392459019087255, + "warnings": [], + "created_at": "2026-09-12T06:44:08.273988" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-8b45bf68-ef1c-4ee8-a362-b5e0f0731221.json b/apps/web/public/demo/evidence-8b45bf68-ef1c-4ee8-a362-b5e0f0731221.json new file mode 100644 index 0000000..54a03ef --- /dev/null +++ b/apps/web/public/demo/evidence-8b45bf68-ef1c-4ee8-a362-b5e0f0731221.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "8b45bf68-ef1c-4ee8-a362-b5e0f0731221", + "tool_name": "compare_periods", + "params": { + "metric": "orders", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 6625.0, + "percent_change": -0.078453192377243, + "absolute_change": -564.0, + "comparison_value": 7189.0 + } + ], + "row_count": 1, + "execution_ms": 12.97525002155453, + "warnings": [], + "created_at": "2026-09-12T06:31:56.550876" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-98c1f67a-6c49-4bd1-b295-65a46065d6b9.json b/apps/web/public/demo/evidence-98c1f67a-6c49-4bd1-b295-65a46065d6b9.json new file mode 100644 index 0000000..addfe35 --- /dev/null +++ b/apps/web/public/demo/evidence-98c1f67a-6c49-4bd1-b295-65a46065d6b9.json @@ -0,0 +1,79 @@ +{ + "evidence_id": "98c1f67a-6c49-4bd1-b295-65a46065d6b9", + "tool_name": "calculate_contribution", + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "rows": [ + { + "change": 94255.10999999999, + "segment_value": "SP", + "share_of_total_change": 0.4639166717936728 + }, + { + "change": 21365.86, + "segment_value": "MG", + "share_of_total_change": 0.10516118077003532 + }, + { + "change": 14713.57, + "segment_value": "PR", + "share_of_total_change": 0.07241910199461049 + }, + { + "change": 14556.739999999998, + "segment_value": "SC", + "share_of_total_change": 0.07164719634793093 + }, + { + "change": 8348.52, + "segment_value": "BA", + "share_of_total_change": 0.04109079722895569 + }, + { + "change": 7488.769999999999, + "segment_value": "ES", + "share_of_total_change": 0.03685917139376638 + }, + { + "change": 6303.230000000003, + "segment_value": "RS", + "share_of_total_change": 0.031024031303449055 + }, + { + "change": 4692.51, + "segment_value": "MS", + "share_of_total_change": 0.02309618673787053 + }, + { + "change": 4077.25, + "segment_value": "PI", + "share_of_total_change": 0.020067922578104812 + }, + { + "change": 3632.0699999999924, + "segment_value": "RJ", + "share_of_total_change": 0.017876779583851123 + } + ], + "row_count": 10, + "execution_ms": 165.86787503911182, + "warnings": [], + "created_at": "2026-09-12T06:43:11.152446" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-c52f6ec6-0cca-48bf-8f61-321e23c9c9d9.json b/apps/web/public/demo/evidence-c52f6ec6-0cca-48bf-8f61-321e23c9c9d9.json new file mode 100644 index 0000000..f0c840b --- /dev/null +++ b/apps/web/public/demo/evidence-c52f6ec6-0cca-48bf-8f61-321e23c9c9d9.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9", + "tool_name": "compare_periods", + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 945606.29, + "percent_change": 0.2736573991331941, + "absolute_change": 203172.5, + "comparison_value": 742433.79 + } + ], + "row_count": 1, + "execution_ms": 230.5426670354791, + "warnings": [], + "created_at": "2026-09-12T06:43:05.643544" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-e2760221-e350-403f-a66b-a189bee1c1b2.json b/apps/web/public/demo/evidence-e2760221-e350-403f-a66b-a189bee1c1b2.json new file mode 100644 index 0000000..23c26cd --- /dev/null +++ b/apps/web/public/demo/evidence-e2760221-e350-403f-a66b-a189bee1c1b2.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "e2760221-e350-403f-a66b-a189bee1c1b2", + "tool_name": "compare_periods", + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 837895.43, + "percent_change": -0.11390666616652897, + "absolute_change": -107710.85999999999, + "comparison_value": 945606.29 + } + ], + "row_count": 1, + "execution_ms": 50.443791958969086, + "warnings": [], + "created_at": "2026-09-12T06:31:56.550876" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-e6ba043a-c288-49db-878a-a7f0720be4ce.json b/apps/web/public/demo/evidence-e6ba043a-c288-49db-878a-a7f0720be4ce.json new file mode 100644 index 0000000..d676a30 --- /dev/null +++ b/apps/web/public/demo/evidence-e6ba043a-c288-49db-878a-a7f0720be4ce.json @@ -0,0 +1,34 @@ +{ + "evidence_id": "e6ba043a-c288-49db-878a-a7f0720be4ce", + "tool_name": "compare_periods", + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "rows": [ + { + "current_value": 993592.98, + "percent_change": 0.012784166402103399, + "absolute_change": 12541.919999999925, + "comparison_value": 981051.06 + } + ], + "row_count": 1, + "execution_ms": 54.635750013403594, + "warnings": [], + "created_at": "2026-09-12T06:44:08.186147" +} \ No newline at end of file diff --git a/apps/web/public/demo/evidence-ede03eec-8ac5-40b6-a940-3cd9c3b44779.json b/apps/web/public/demo/evidence-ede03eec-8ac5-40b6-a940-3cd9c3b44779.json new file mode 100644 index 0000000..bb96491 --- /dev/null +++ b/apps/web/public/demo/evidence-ede03eec-8ac5-40b6-a940-3cd9c3b44779.json @@ -0,0 +1,79 @@ +{ + "evidence_id": "ede03eec-8ac5-40b6-a940-3cd9c3b44779", + "tool_name": "calculate_contribution", + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "rows": [ + { + "change": -64619.669999999984, + "segment_value": "SP", + "share_of_total_change": 0.5999364409494083 + }, + { + "change": -7155.149999999994, + "segment_value": "MG", + "share_of_total_change": 0.06642923471226574 + }, + { + "change": -6835.479999999996, + "segment_value": "SC", + "share_of_total_change": 0.0634613817028292 + }, + { + "change": -5393.240000000005, + "segment_value": "RJ", + "share_of_total_change": 0.05007145983236979 + }, + { + "change": -5044.26, + "segment_value": "PA", + "share_of_total_change": 0.04683148941527345 + }, + { + "change": -4749.4000000000015, + "segment_value": "BA", + "share_of_total_change": 0.044093975296455735 + }, + { + "change": -3962.010000000001, + "segment_value": "MT", + "share_of_total_change": 0.036783756066937 + }, + { + "change": 3468.41, + "segment_value": "DF", + "share_of_total_change": -0.032201116953295146 + }, + { + "change": -3101.2899999999972, + "segment_value": "ES", + "share_of_total_change": 0.028792732691949516 + }, + { + "change": -2603.0699999999997, + "segment_value": "RN", + "share_of_total_change": 0.024167200967479045 + } + ], + "row_count": 10, + "execution_ms": 64.36770898289979, + "warnings": [], + "created_at": "2026-09-12T06:32:00.140816" +} \ No newline at end of file diff --git a/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70-events.json b/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70-events.json new file mode 100644 index 0000000..de79d6b --- /dev/null +++ b/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70-events.json @@ -0,0 +1,247 @@ +[ + { + "id": 183, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 837895.43, + "percent_change": -0.11390666616652897, + "absolute_change": -107710.85999999999, + "comparison_value": 945606.29 + } + ], + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "e2760221-e350-403f-a66b-a189bee1c1b2", + "execution_ms": 50.443791958969086 + }, + "created_at": "2026-09-12T06:31:56.550876" + }, + { + "id": 184, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 6625.0, + "percent_change": -0.078453192377243, + "absolute_change": -564.0, + "comparison_value": 7189.0 + } + ], + "params": { + "metric": "orders", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "8b45bf68-ef1c-4ee8-a362-b5e0f0731221", + "execution_ms": 12.97525002155453 + }, + "created_at": "2026-09-12T06:31:56.550876" + }, + { + "id": 185, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 0.015309155766944114, + "percent_change": 0.3255177368212446, + "absolute_change": 0.003759588875702536, + "comparison_value": 0.011549566891241578 + } + ], + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "00d6a3fb-019f-4952-bbfe-bc46bbdc7106", + "execution_ms": 13.579958991613239 + }, + "created_at": "2026-09-12T06:31:56.633801" + }, + { + "id": 186, + "event_type": "decomposition_plan", + "payload": { + "metric": "cancellation", + "rationale": "The cancellation rate increased by 32.6%, significantly higher than the 7.8% decrease in order count. Since order count already excludes cancelled orders, the sharp rise in cancellation rate indicates that the drop in product_revenue is primarily driven by cancellations rather than a genuine decline in demand.", + "primary_driver_hypothesis": "cancellation", + "recommended_next_dimension": "customer_state" + }, + "created_at": "2026-09-12T06:32:00.065976" + }, + { + "id": 187, + "event_type": "hypothesis_update", + "payload": { + "id": "7eacea4f-5fd0-4f12-a355-418e91dfa98d", + "status": "testing", + "statement": "The change in product_revenue is primarily driven by cancellation (The cancellation rate increased by 32.6%, significantly higher than the 7.8% decrease in order count. Since order count already excludes cancelled orders, the sharp rise in cancellation rate indicates that the drop in product_revenue is primarily driven by cancellations rather than a genuine decline in demand.)", + "confidence": null, + "supporting_evidence_ids": [], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:32:00.065976" + }, + { + "id": 188, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "rows": [ + { + "change": -64619.669999999984, + "segment_value": "SP", + "share_of_total_change": 0.5999364409494083 + }, + { + "change": -7155.149999999994, + "segment_value": "MG", + "share_of_total_change": 0.06642923471226574 + }, + { + "change": -6835.479999999996, + "segment_value": "SC", + "share_of_total_change": 0.0634613817028292 + }, + { + "change": -5393.240000000005, + "segment_value": "RJ", + "share_of_total_change": 0.05007145983236979 + }, + { + "change": -5044.26, + "segment_value": "PA", + "share_of_total_change": 0.04683148941527345 + }, + { + "change": -4749.4000000000015, + "segment_value": "BA", + "share_of_total_change": 0.044093975296455735 + }, + { + "change": -3962.010000000001, + "segment_value": "MT", + "share_of_total_change": 0.036783756066937 + }, + { + "change": 3468.41, + "segment_value": "DF", + "share_of_total_change": -0.032201116953295146 + }, + { + "change": -3101.2899999999972, + "segment_value": "ES", + "share_of_total_change": 0.028792732691949516 + }, + { + "change": -2603.0699999999997, + "segment_value": "RN", + "share_of_total_change": 0.024167200967479045 + } + ], + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-02-28", + "start": "2018-02-01" + }, + "comparison_period": { + "end": "2018-01-31", + "start": "2018-01-01" + } + }, + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "warnings": [], + "row_count": 10, + "tool_name": "calculate_contribution", + "evidence_id": "ede03eec-8ac5-40b6-a940-3cd9c3b44779", + "execution_ms": 64.36770898289979 + }, + "created_at": "2026-09-12T06:32:00.140816" + }, + { + "id": 189, + "event_type": "hypothesis_update", + "payload": { + "id": "7eacea4f-5fd0-4f12-a355-418e91dfa98d", + "status": "supported", + "statement": "The change in product_revenue is primarily driven by cancellation (The cancellation rate increased by 32.6%, significantly higher than the 7.8% decrease in order count. Since order count already excludes cancelled orders, the sharp rise in cancellation rate indicates that the drop in product_revenue is primarily driven by cancellations rather than a genuine decline in demand.)", + "confidence": "medium", + "supporting_evidence_ids": [ + "ede03eec-8ac5-40b6-a940-3cd9c3b44779" + ], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:32:00.140816" + }, + { + "id": 190, + "event_type": "report_generated", + "payload": { + "status": "answered", + "headline": "Product Revenue Decline Driven by Cancellations in SP Segment" + }, + "created_at": "2026-09-12T06:32:36.325011" + } +] \ No newline at end of file diff --git a/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70.json b/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70.json new file mode 100644 index 0000000..e288b79 --- /dev/null +++ b/apps/web/public/demo/investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70.json @@ -0,0 +1,75 @@ +{ + "investigation_id": "17f4ecd2-a471-4b18-84ab-911d90b7fe70", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-02-01", + "end": "2018-02-28" + }, + "comparison_period": { + "start": "2018-01-01", + "end": "2018-01-31" + }, + "question": null, + "status": "completed", + "step_count": 5, + "query_count": 4, + "cancel_requested": false, + "created_at": "2026-09-12T06:31:56.516588", + "updated_at": "2026-09-12T06:32:36.325011", + "report": { + "status": "answered", + "headline": "Product Revenue Decline Driven by Cancellations in SP Segment", + "observed_change": { + "metric": "product_revenue", + "current_value": 837895.43, + "comparison_value": 945606.29, + "percent_change": -0.11390666616652897, + "evidence_ids": [ + "e2760221-e350-403f-a66b-a189bee1c1b2" + ] + }, + "findings": [ + { + "claim": "Product revenue decreased by 18.2% in the period, with the SP segment contributing 60.0% of the total decline.", + "claim_type": "interpretation", + "evidence_ids": [ + "00d6a3fb-019f-4952-bbfe-bc46bbdc7106" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate increased by 32.6% in the period, outpacing the 7.8% decrease in order count.", + "claim_type": "interpretation", + "evidence_ids": [], + "confidence": "high" + }, + { + "claim": "The SP segment accounted for 60.0% of the total change in product_revenue.", + "claim_type": "interpretation", + "evidence_ids": [ + "e2760221-e350-403f-a66b-a189bee1c1b2" + ], + "confidence": "high" + }, + { + "claim": "Order count decreased by 7.8%, while the cancellation rate increased by 32.6%.", + "claim_type": "interpretation", + "evidence_ids": [ + "ede03eec-8ac5-40b6-a940-3cd9c3b44779" + ], + "confidence": "high" + } + ], + "limitations": [ + "The analysis does not account for external factors such as market trends or competitor activity.", + "The data does not include information on customer behavior changes or product quality issues.", + "The relationship between cancellation rate and revenue decline is inferred but not directly measured." + ], + "recommended_next_checks": [ + "Review the SP segment's customer feedback and support tickets for patterns in cancellations.", + "Analyze the time series of cancellation rates and order counts to identify any seasonal trends or anomalies.", + "Conduct a cohort analysis of SP segment customers to assess retention and churn behavior over time." + ] + } +} \ No newline at end of file diff --git a/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352-events.json b/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352-events.json new file mode 100644 index 0000000..7033677 --- /dev/null +++ b/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352-events.json @@ -0,0 +1,275 @@ +[ + { + "id": 197, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 945606.29, + "percent_change": 0.2736573991331941, + "absolute_change": 203172.5, + "comparison_value": 742433.79 + } + ], + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9", + "execution_ms": 230.5426670354791 + }, + "created_at": "2026-09-12T06:43:05.643544" + }, + { + "id": 198, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 7189.0, + "percent_change": 0.27849902187444425, + "absolute_change": 1566.0, + "comparison_value": 5623.0 + } + ], + "params": { + "metric": "orders", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "3be66378-ed39-4865-92f1-2e8a7c83bca7", + "execution_ms": 11.721291986759752 + }, + "created_at": "2026-09-12T06:43:05.643544" + }, + { + "id": 199, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 0.011549566891241578, + "percent_change": 0.21420168965886, + "absolute_change": 0.0020375006590767023, + "comparison_value": 0.009512066232164875 + } + ], + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "6d536dc5-14a1-44e9-803a-3966949e6d3f", + "execution_ms": 11.227999988477677 + }, + "created_at": "2026-09-12T06:43:05.910937" + }, + { + "id": 200, + "event_type": "decomposition_plan", + "payload": { + "metric": "product_revenue", + "rationale": "The product_revenue increased by 27.4% while order count increased by 27.8% (nearly identical percentage change). However, the cancellation rate rose by 21.4%, indicating a higher proportion of orders being cancelled. Since order count already excludes cancelled orders, the cancellation rate increase suggests that the revenue growth may be partially offset by cancellations, making cancellation the primary driver.", + "primary_driver_hypothesis": "cancellation", + "recommended_next_dimension": "customer_state" + }, + "created_at": "2026-09-12T06:43:10.967399" + }, + { + "id": 201, + "event_type": "hypothesis_update", + "payload": { + "id": "bc194979-c238-4182-8251-604183f73f6e", + "status": "testing", + "statement": "The change in product_revenue is primarily driven by cancellation (The product_revenue increased by 27.4% while order count increased by 27.8% (nearly identical percentage change). However, the cancellation rate rose by 21.4%, indicating a higher proportion of orders being cancelled. Since order count already excludes cancelled orders, the cancellation rate increase suggests that the revenue growth may be partially offset by cancellations, making cancellation the primary driver.)", + "confidence": null, + "supporting_evidence_ids": [], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:43:10.967399" + }, + { + "id": 202, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "rows": [ + { + "change": 94255.10999999999, + "segment_value": "SP", + "share_of_total_change": 0.4639166717936728 + }, + { + "change": 21365.86, + "segment_value": "MG", + "share_of_total_change": 0.10516118077003532 + }, + { + "change": 14713.57, + "segment_value": "PR", + "share_of_total_change": 0.07241910199461049 + }, + { + "change": 14556.739999999998, + "segment_value": "SC", + "share_of_total_change": 0.07164719634793093 + }, + { + "change": 8348.52, + "segment_value": "BA", + "share_of_total_change": 0.04109079722895569 + }, + { + "change": 7488.769999999999, + "segment_value": "ES", + "share_of_total_change": 0.03685917139376638 + }, + { + "change": 6303.230000000003, + "segment_value": "RS", + "share_of_total_change": 0.031024031303449055 + }, + { + "change": 4692.51, + "segment_value": "MS", + "share_of_total_change": 0.02309618673787053 + }, + { + "change": 4077.25, + "segment_value": "PI", + "share_of_total_change": 0.020067922578104812 + }, + { + "change": 3632.0699999999924, + "segment_value": "RJ", + "share_of_total_change": 0.017876779583851123 + } + ], + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-01-31", + "start": "2018-01-01" + }, + "comparison_period": { + "end": "2017-12-31", + "start": "2017-12-01" + } + }, + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "warnings": [], + "row_count": 10, + "tool_name": "calculate_contribution", + "evidence_id": "98c1f67a-6c49-4bd1-b295-65a46065d6b9", + "execution_ms": 165.86787503911182 + }, + "created_at": "2026-09-12T06:43:11.152446" + }, + { + "id": 203, + "event_type": "hypothesis_update", + "payload": { + "id": "bc194979-c238-4182-8251-604183f73f6e", + "status": "inconclusive", + "statement": "The change in product_revenue is primarily driven by cancellation (The product_revenue increased by 27.4% while order count increased by 27.8% (nearly identical percentage change). However, the cancellation rate rose by 21.4%, indicating a higher proportion of orders being cancelled. Since order count already excludes cancelled orders, the cancellation rate increase suggests that the revenue growth may be partially offset by cancellations, making cancellation the primary driver.)", + "confidence": "low", + "supporting_evidence_ids": [], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:43:11.152446" + }, + { + "id": 204, + "event_type": "ad_hoc_query_decision", + "payload": { + "query": "SELECT \n SUM(CASE WHEN o.order_status = 'delivered' THEN 1 ELSE 0 END) AS delivered_orders,\n SUM(CASE WHEN o.order_status = 'canceled' THEN 1 ELSE 0 END) AS canceled_orders,\n SUM(CASE WHEN o.order_status = 'unavailable' THEN 1 ELSE 0 END) AS unavailable_orders,\n COUNT(*) AS total_orders,\n SUM(CASE WHEN o.order_status = 'delivered' THEN oi.price ELSE 0 END) AS delivered_revenue,\n SUM(CASE WHEN o.order_status = 'canceled' THEN oi.price ELSE 0 END) AS canceled_revenue,\n SUM(CASE WHEN o.order_status = 'unavailable' THEN oi.price ELSE 0 END) AS unavailable_revenue,\n SUM(oi.price) AS total_revenue\nFROM orders o\nJOIN order_items oi ON o.order_id = oi.order_id\nWHERE o.order_purchase_timestamp BETWEEN '2017-12-01' AND '2017-12-31'\n OR o.order_purchase_timestamp BETWEEN '2018-01-01' AND '2018-01-31'\nORDER BY o.order_purchase_timestamp;", + "purpose": "To determine the proportion of revenue from delivered orders versus canceled orders in each period, which will clarify whether the revenue growth is due to more orders being fulfilled or due to higher cancellation rates.", + "should_query": true + }, + "created_at": "2026-09-12T06:43:20.669838" + }, + { + "id": 205, + "event_type": "ad_hoc_query_rejected", + "payload": { + "query": "SELECT \n SUM(CASE WHEN o.order_status = 'delivered' THEN 1 ELSE 0 END) AS delivered_orders,\n SUM(CASE WHEN o.order_status = 'canceled' THEN 1 ELSE 0 END) AS canceled_orders,\n SUM(CASE WHEN o.order_status = 'unavailable' THEN 1 ELSE 0 END) AS unavailable_orders,\n COUNT(*) AS total_orders,\n SUM(CASE WHEN o.order_status = 'delivered' THEN oi.price ELSE 0 END) AS delivered_revenue,\n SUM(CASE WHEN o.order_status = 'canceled' THEN oi.price ELSE 0 END) AS canceled_revenue,\n SUM(CASE WHEN o.order_status = 'unavailable' THEN oi.price ELSE 0 END) AS unavailable_revenue,\n SUM(oi.price) AS total_revenue\nFROM orders o\nJOIN order_items oi ON o.order_id = oi.order_id\nWHERE o.order_purchase_timestamp BETWEEN '2017-12-01' AND '2017-12-31'\n OR o.order_purchase_timestamp BETWEEN '2018-01-01' AND '2018-01-31'\nORDER BY o.order_purchase_timestamp;", + "reason": "column \"o.order_purchase_timestamp\" must appear in the GROUP BY clause or be used in an aggregate function\nLINE 1: ...mp BETWEEN '2018-01-01' AND '2018-01-31' ORDER BY o.order_pu...\n ^", + "attempt": 1 + }, + "created_at": "2026-09-12T06:43:20.706464" + }, + { + "id": 206, + "event_type": "ad_hoc_query_rejected", + "payload": { + "query": "WITH current_period AS ( SELECT COUNT(DISTINCT o.order_id) AS current_order_count, SUM(oi.price * oi.order_item_id) AS current_revenue, SUM(oi.price * oi.order_item_id) AS current_revenue, SUM(oi.price * oi.order_item_id) AS current_revenue FROM orders o JOIN order_items oi ON o.order_id = oi.order_id WHERE o.order_purchase_timestamp BETWEEN '2018-01-01' AND '2018-01-31' AND o.order_status = 'delivered' ), comparison_period AS ( SELECT COUNT(DISTINCT o.order_id) AS comparison_order_count, SUM(oi.price * oi.order_item_id) AS comparison_revenue, SUM(oi.price * oi.order_item_id) AS comparison_revenue, SUM(oi.price * oi.order_item_id) AS comparison_revenue FROM orders o JOIN order_items oi ON o.order_id = oi.order_id WHERE o.order_purchase_timestamp BETWEEN '2017-12-01' AND '2017-12-31' AND o.order_status = 'delivered' ), revenue_growth AS ( SELECT (current_revenue / comparison_revenue - 1) * 100 AS revenue_growth_percent, (current_order_count / comparison_order_count - 1) * 100 AS order_count_growth_percent FROM current_period, comparison_period ), cancellation_rate AS ( SELECT (SUM(CASE WHEN o.order_status = 'cancelled' THEN 1 ELSE 0 END) / COUNT(*)) * 100 AS cancellation_rate FROM orders o WHERE o.order_purchase_timestamp BETWEEN '2017-12-01' AND '2018-01-31' AND o.order_status IN ('cancelled', 'delivered') ) SELECT * FROM revenue_growth, cancellation_rate;", + "reason": "column reference \"current_revenue\" is ambiguous\nLINE 1: ...status = 'delivered'), revenue_growth AS (SELECT (current_re...\n ^", + "attempt": 2 + }, + "created_at": "2026-09-12T06:43:31.292374" + }, + { + "id": 207, + "event_type": "report_generated", + "payload": { + "status": "partial", + "headline": "Product Revenue Growth Driven by SP Segment but Cancellation Rate Increase Raises Concerns" + }, + "created_at": "2026-09-12T06:44:06.908283" + } +] \ No newline at end of file diff --git a/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352.json b/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352.json new file mode 100644 index 0000000..d47c620 --- /dev/null +++ b/apps/web/public/demo/investigation-69182461-d134-401a-a2d9-566f5125c352.json @@ -0,0 +1,85 @@ +{ + "investigation_id": "69182461-d134-401a-a2d9-566f5125c352", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-01-01", + "end": "2018-01-31" + }, + "comparison_period": { + "start": "2017-12-01", + "end": "2017-12-31" + }, + "question": "Why did revenue decline?", + "status": "completed", + "step_count": 8, + "query_count": 6, + "cancel_requested": false, + "created_at": "2026-09-12T06:43:05.610388", + "updated_at": "2026-09-12T06:44:06.908283", + "report": { + "status": "partial", + "headline": "Product Revenue Growth Driven by SP Segment but Cancellation Rate Increase Raises Concerns", + "observed_change": { + "metric": "product_revenue", + "current_value": 945606.29, + "comparison_value": 742433.79, + "percent_change": 0.2736573991331941, + "evidence_ids": [ + "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9" + ] + }, + "findings": [ + { + "claim": "Product_revenue increased by 27.4% (evidence ID: 3be66378-ed39-4865-92f1-2e8a7c83bca7)", + "claim_type": "interpretation", + "evidence_ids": [ + "3be66378-ed39-4865-92f1-2e8a7c83bca7" + ], + "confidence": "high" + }, + { + "claim": "Order count increased by 27.8% (evidence ID: 6d536dc5-14a1-44e9-803a-3966949e6d3f)", + "claim_type": "interpretation", + "evidence_ids": [ + "6d536dc5-14a1-44e9-803a-3966949e6d3f" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate rose by 21.4% (evidence ID: 98c1f67a-6c49-4bd1-b295-65a46065d6b9)", + "claim_type": "interpretation", + "evidence_ids": [ + "98c1f67a-6c49-4bd1-b295-65a46065d6b9" + ], + "confidence": "high" + }, + { + "claim": "The SP segment contributed 46.4% of the total change (evidence ID: c52f6ec6-0cca-48bf-8f61-321e23c9c9d9)", + "claim_type": "interpretation", + "evidence_ids": [ + "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9" + ], + "confidence": "high" + }, + { + "claim": "The cancellation rate increase suggests that revenue growth may be partially offset by cancellations (evidence ID: 98c1f67a-6c49-4bd1-b295-65a46065d6b9)", + "claim_type": "interpretation", + "evidence_ids": [ + "98c1f67a-6c49-4bd1-b295-65a46065d6b9" + ], + "confidence": "medium" + } + ], + "limitations": [ + "The hypothesis is inconclusive due to limited data on the relationship between cancellation rates and revenue growth.", + "The analysis does not account for external factors such as market trends or competitor activity.", + "The contribution of the SP segment to revenue growth is not fully validated without further analysis." + ], + "recommended_next_checks": [ + "Investigate the correlation between cancellation rates and revenue growth using detailed transaction data.", + "Analyze the SP segment's performance to validate its contribution to revenue growth.", + "Conduct a deeper analysis of the reasons behind the increased cancellation rate." + ] + } +} \ No newline at end of file diff --git a/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675-events.json b/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675-events.json new file mode 100644 index 0000000..74980c3 --- /dev/null +++ b/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675-events.json @@ -0,0 +1,247 @@ +[ + { + "id": 208, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 993592.98, + "percent_change": 0.012784166402103399, + "absolute_change": 12541.919999999925, + "comparison_value": 981051.06 + } + ], + "params": { + "metric": "product_revenue", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "e6ba043a-c288-49db-878a-a7f0720be4ce", + "execution_ms": 54.635750013403594 + }, + "created_at": "2026-09-12T06:44:08.186147" + }, + { + "id": 209, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 6919.0, + "percent_change": -0.03473772321428571, + "absolute_change": -249.0, + "comparison_value": 7168.0 + } + ], + "params": { + "metric": "orders", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "7764dd7b-f478-4bde-8610-64ca1e93dc35", + "execution_ms": 13.631290988996625 + }, + "created_at": "2026-09-12T06:44:08.186147" + }, + { + "id": 210, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.compare_periods (parameterized query, not a raw SQL string)", + "rows": [ + { + "current_value": 0.002882259691598213, + "percent_change": -0.5166517526484949, + "absolute_change": -0.003080852220757909, + "comparison_value": 0.005963111912356122 + } + ], + "params": { + "metric": "cancellation_rate", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "columns": [ + "current_value", + "comparison_value", + "absolute_change", + "percent_change" + ], + "warnings": [], + "row_count": 1, + "tool_name": "compare_periods", + "evidence_id": "8595a0fb-6e94-4f97-b610-67cde1cf001e", + "execution_ms": 13.392459019087255 + }, + "created_at": "2026-09-12T06:44:08.273988" + }, + { + "id": 211, + "event_type": "decomposition_plan", + "payload": { + "metric": "product_revenue", + "rationale": "product_revenue increased by 1.3% while order count decreased by 3.5% and cancellation rate dropped by 51.7%. The significant drop in cancellation rate suggests that the decrease in order count was not due to cancellations but rather a genuine decline in demand. However, the revenue increase indicates that the remaining orders had higher average order values. Since the cancellation rate fell sharply, the primary driver is not cancellation but a mix of factors. However, the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. Therefore, the primary driver is a mix, but based on the given choices, the most fitting is 'cancellation' due to the sharp drop in cancellation rate. Wait, the cancellation rate actually decreased, not increased. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. However, the revenue increased despite fewer orders, implying higher average order value. The question asks to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. But the revenue increased, so average order value may have increased. However, the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. However, the options are cancellation, order_volume, or a mix. So the answer is a mix. But the user might expect a different answer. Let me recheck. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the options are cancellation, order_volume, or a mix. So the answer is a mix. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume. But the revenue increase implies average order value. So the primary driver is a mix. Therefore, the primary driver is a mix. The next dimension to inspect is customer_state, product_category, or seller. Since the revenue increased despite fewer orders, average order value is a factor. To determine if it's due to customer_state, product_category, or seller, we need to inspect these dimensions. However, the question asks to recommend which dimension to inspect next. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. The question says to recommend which dimension to inspect next. Since the revenue increased, it's likely due to average order value, which could be influenced by customer_state, product_category, or seller. So the next step is to inspect customer_state, product_category, or seller. But the question says to choose between the three. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. But the question says to recommend which dimension to inspect next. Since the primary driver is a mix, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. However, the question says to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. The answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too convoluted. Let me go back to the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the final answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is confusing. Let me recheck the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too complicated. Let me go back to the initial instruction. The user says: ", + "primary_driver_hypothesis": "cancellation", + "recommended_next_dimension": "customer_state" + }, + "created_at": "2026-09-12T06:45:02.374381" + }, + { + "id": 212, + "event_type": "hypothesis_update", + "payload": { + "id": "bacb8d13-e540-42d4-af3c-8d48f9b73fdb", + "status": "testing", + "statement": "The change in product_revenue is primarily driven by cancellation (product_revenue increased by 1.3% while order count decreased by 3.5% and cancellation rate dropped by 51.7%. The significant drop in cancellation rate suggests that the decrease in order count was not due to cancellations but rather a genuine decline in demand. However, the revenue increase indicates that the remaining orders had higher average order values. Since the cancellation rate fell sharply, the primary driver is not cancellation but a mix of factors. However, the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. Therefore, the primary driver is a mix, but based on the given choices, the most fitting is 'cancellation' due to the sharp drop in cancellation rate. Wait, the cancellation rate actually decreased, not increased. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. However, the revenue increased despite fewer orders, implying higher average order value. The question asks to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. But the revenue increased, so average order value may have increased. However, the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. However, the options are cancellation, order_volume, or a mix. So the answer is a mix. But the user might expect a different answer. Let me recheck. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the options are cancellation, order_volume, or a mix. So the answer is a mix. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume. But the revenue increase implies average order value. So the primary driver is a mix. Therefore, the primary driver is a mix. The next dimension to inspect is customer_state, product_category, or seller. Since the revenue increased despite fewer orders, average order value is a factor. To determine if it's due to customer_state, product_category, or seller, we need to inspect these dimensions. However, the question asks to recommend which dimension to inspect next. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. The question says to recommend which dimension to inspect next. Since the revenue increased, it's likely due to average order value, which could be influenced by customer_state, product_category, or seller. So the next step is to inspect customer_state, product_category, or seller. But the question says to choose between the three. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. But the question says to recommend which dimension to inspect next. Since the primary driver is a mix, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. However, the question says to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. The answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too convoluted. Let me go back to the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the final answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is confusing. Let me recheck the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too complicated. Let me go back to the initial instruction. The user says: )", + "confidence": null, + "supporting_evidence_ids": [], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:45:02.374381" + }, + { + "id": 213, + "event_type": "tool_call", + "payload": { + "sql": "see app.analytics.calculate_contribution (derived from segment_metric queries)", + "rows": [ + { + "change": 15359.749999999993, + "segment_value": "PR", + "share_of_total_change": 1.2246729368390235 + }, + { + "change": 12794.570000000007, + "segment_value": "SP", + "share_of_total_change": 1.02014444359397 + }, + { + "change": -8453.86, + "segment_value": "MG", + "share_of_total_change": -0.6740483115822817 + }, + { + "change": 7940.32, + "segment_value": "SC", + "share_of_total_change": 0.6331024276984741 + }, + { + "change": 5432.899999999994, + "segment_value": "RJ", + "share_of_total_change": 0.43317928993328186 + }, + { + "change": -5198.779999999999, + "segment_value": "PE", + "share_of_total_change": -0.4145122915789632 + }, + { + "change": -5058.630000000005, + "segment_value": "RS", + "share_of_total_change": -0.40333776646638114 + }, + { + "change": -4872.84, + "segment_value": "BA", + "share_of_total_change": -0.38852424509166295 + }, + { + "change": 4217.64, + "segment_value": "PI", + "share_of_total_change": 0.3362834398561006 + }, + { + "change": 2905.66, + "segment_value": "MT", + "share_of_total_change": 0.23167585186319298 + } + ], + "params": { + "limit": 10, + "metric": "product_revenue", + "dimension": "customer_state", + "current_period": { + "end": "2018-04-30", + "start": "2018-04-01" + }, + "comparison_period": { + "end": "2018-03-31", + "start": "2018-03-01" + } + }, + "columns": [ + "segment_value", + "change", + "share_of_total_change" + ], + "warnings": [], + "row_count": 10, + "tool_name": "calculate_contribution", + "evidence_id": "0d5a34b0-4b86-48a9-aa49-887d234cda61", + "execution_ms": 70.6508329603821 + }, + "created_at": "2026-09-12T06:45:02.463282" + }, + { + "id": 214, + "event_type": "hypothesis_update", + "payload": { + "id": "bacb8d13-e540-42d4-af3c-8d48f9b73fdb", + "status": "supported", + "statement": "The change in product_revenue is primarily driven by cancellation (product_revenue increased by 1.3% while order count decreased by 3.5% and cancellation rate dropped by 51.7%. The significant drop in cancellation rate suggests that the decrease in order count was not due to cancellations but rather a genuine decline in demand. However, the revenue increase indicates that the remaining orders had higher average order values. Since the cancellation rate fell sharply, the primary driver is not cancellation but a mix of factors. However, the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. Therefore, the primary driver is a mix, but based on the given choices, the most fitting is 'cancellation' due to the sharp drop in cancellation rate. Wait, the cancellation rate actually decreased, not increased. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. However, the revenue increased despite fewer orders, implying higher average order value. The question asks to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The decrease in order count without a corresponding rise in cancellation rate suggests the primary driver is order_volume. But the revenue increased, so average order value may have increased. However, the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question specifies to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. Since the cancellation rate decreased, the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. However, the options are cancellation, order_volume, or a mix. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. However, the options are cancellation, order_volume, or a mix. So the answer is a mix. But the user might expect a different answer. Let me recheck. The question says to choose 'cancellation' when cancellation rate rose sharply, which it did not. So the primary driver is not cancellation. The order count decreased, so order_volume is a factor. The revenue increased, so average order value is also a factor. Therefore, the primary driver is a mix. But the options are cancellation, order_volume, or a mix. So the answer is a mix. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume. But the revenue increase implies average order value. So the primary driver is a mix. Therefore, the primary driver is a mix. The next dimension to inspect is customer_state, product_category, or seller. Since the revenue increased despite fewer orders, average order value is a factor. To determine if it's due to customer_state, product_category, or seller, we need to inspect these dimensions. However, the question asks to recommend which dimension to inspect next. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. The question says to recommend which dimension to inspect next. Since the revenue increased, it's likely due to average order value, which could be influenced by customer_state, product_category, or seller. So the next step is to inspect customer_state, product_category, or seller. But the question says to choose between the three. Since the primary driver is a mix, but the options are to choose between cancellation, order_volume, or a mix, and the correct answer is a mix, the next step is to inspect customer_state, product_category, or seller. However, the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. But the question says to recommend which dimension to inspect next. Since the primary driver is a mix, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. However, the question says to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the decrease in order count without a rise in cancellation rate suggests order_volume, and the revenue increase suggests average order value. Therefore, the primary driver is a mix, and the next step is to inspect customer_state, product_category, or seller. The answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too convoluted. Let me go back to the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the final answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is confusing. Let me recheck the initial data. product_revenue increased by 1.3%, order count decreased by 3.5%, cancellation rate decreased by 51.7%. The cancellation rate decreased, so the decrease in order count was not due to cancellations but a genuine decline in demand. However, the revenue increased, implying higher average order value. So the primary driver is a mix of order_volume (fewer orders) and average order value (higher). But the question asks to choose between cancellation, order_volume, or a mix. The correct answer is a mix. Therefore, the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume, and the next step is to inspect customer_state, product_category, or seller. However, the correct answer is a mix. So the answer is: primary driver is a mix, next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. However, the correct answer is a mix. Therefore, the answer is a mix, and the next step is to inspect customer_state, product_category, or seller. But the user might have intended that the primary driver is order_volume. This is getting too complicated. Let me go back to the initial instruction. The user says: )", + "confidence": "high", + "supporting_evidence_ids": [ + "0d5a34b0-4b86-48a9-aa49-887d234cda61" + ], + "contradicting_evidence_ids": [] + }, + "created_at": "2026-09-12T06:45:02.463282" + }, + { + "id": 215, + "event_type": "report_generated", + "payload": { + "status": "answered", + "headline": "Product Revenue Increase Driven by Mix of Order Volume and Average Order Value" + }, + "created_at": "2026-09-12T06:45:38.554105" + } +] \ No newline at end of file diff --git a/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675.json b/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675.json new file mode 100644 index 0000000..c367e77 --- /dev/null +++ b/apps/web/public/demo/investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675.json @@ -0,0 +1,76 @@ +{ + "investigation_id": "f0c3ba87-8f54-495a-b3e5-b8531e064675", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-04-01", + "end": "2018-04-30" + }, + "comparison_period": { + "start": "2018-03-01", + "end": "2018-03-31" + }, + "question": "What drove the change in revenue this period?", + "status": "completed", + "step_count": 5, + "query_count": 4, + "cancel_requested": false, + "created_at": "2026-09-12T06:44:08.162011", + "updated_at": "2026-09-12T06:45:38.554105", + "report": { + "status": "answered", + "headline": "Product Revenue Increase Driven by Mix of Order Volume and Average Order Value", + "observed_change": { + "metric": "product_revenue", + "current_value": 993592.98, + "comparison_value": 981051.06, + "percent_change": 0.012784166402103399, + "evidence_ids": [ + "e6ba043a-c288-49db-878a-a7f0720be4ce" + ] + }, + "findings": [ + { + "claim": "Product_revenue increased by 1.3% (evidence ID: 0d5a34b0-4b86-48a9-aa49-887d234cda61)", + "claim_type": "interpretation", + "evidence_ids": [ + "0d5a34b0-4b86-48a9-aa49-887d234cda61" + ], + "confidence": "high" + }, + { + "claim": "Order count decreased by 3.5% (evidence ID: 7764dd7b-f478-4bde-8610-64ca1e93dc35)", + "claim_type": "interpretation", + "evidence_ids": [ + "7764dd7b-f478-4bde-8610-64ca1e93dc35" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate decreased by 51.7% (evidence ID: 8595a0fb-6e94-4f97-b610-67cde1cf001e)", + "claim_type": "interpretation", + "evidence_ids": [ + "8595a0fb-6e94-4f97-b610-67cde1cf001e" + ], + "confidence": "high" + }, + { + "claim": "The largest contributing segment was 'PR', responsible for 122.5% of the total change (evidence ID: e6ba043a-c288-49db-878a-a7f0720be4ce)", + "claim_type": "interpretation", + "evidence_ids": [ + "e6ba043a-c288-49db-878a-a7f0720be4ce" + ], + "confidence": "high" + } + ], + "limitations": [ + "The data does not specify the exact reasons for the change in average order value", + "The analysis does not account for external factors such as market trends or competitor activity" + ], + "recommended_next_checks": [ + "Inspect customer_state to understand if specific regions contributed to the average order value increase", + "Review product_category to identify if certain product types drove the revenue growth", + "Analyze seller performance to determine if specific sellers influenced the change in average order value" + ] + } +} \ No newline at end of file diff --git a/apps/web/public/demo/investigation-ids.json b/apps/web/public/demo/investigation-ids.json new file mode 100644 index 0000000..b5f6c48 --- /dev/null +++ b/apps/web/public/demo/investigation-ids.json @@ -0,0 +1,5 @@ +[ + "69182461-d134-401a-a2d9-566f5125c352", + "17f4ecd2-a471-4b18-84ab-911d90b7fe70", + "f0c3ba87-8f54-495a-b3e5-b8531e064675" +] \ No newline at end of file diff --git a/apps/web/public/demo/investigations-list.json b/apps/web/public/demo/investigations-list.json new file mode 100644 index 0000000..8a0ced0 --- /dev/null +++ b/apps/web/public/demo/investigations-list.json @@ -0,0 +1,238 @@ +[ + { + "investigation_id": "f0c3ba87-8f54-495a-b3e5-b8531e064675", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-04-01", + "end": "2018-04-30" + }, + "comparison_period": { + "start": "2018-03-01", + "end": "2018-03-31" + }, + "question": "What drove the change in revenue this period?", + "status": "completed", + "step_count": 5, + "query_count": 4, + "cancel_requested": false, + "created_at": "2026-09-12T06:44:08.162011", + "updated_at": "2026-09-12T06:45:38.554105", + "report": { + "status": "answered", + "headline": "Product Revenue Increase Driven by Mix of Order Volume and Average Order Value", + "observed_change": { + "metric": "product_revenue", + "current_value": 993592.98, + "comparison_value": 981051.06, + "percent_change": 0.012784166402103399, + "evidence_ids": [ + "e6ba043a-c288-49db-878a-a7f0720be4ce" + ] + }, + "findings": [ + { + "claim": "Product_revenue increased by 1.3% (evidence ID: 0d5a34b0-4b86-48a9-aa49-887d234cda61)", + "claim_type": "interpretation", + "evidence_ids": [ + "0d5a34b0-4b86-48a9-aa49-887d234cda61" + ], + "confidence": "high" + }, + { + "claim": "Order count decreased by 3.5% (evidence ID: 7764dd7b-f478-4bde-8610-64ca1e93dc35)", + "claim_type": "interpretation", + "evidence_ids": [ + "7764dd7b-f478-4bde-8610-64ca1e93dc35" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate decreased by 51.7% (evidence ID: 8595a0fb-6e94-4f97-b610-67cde1cf001e)", + "claim_type": "interpretation", + "evidence_ids": [ + "8595a0fb-6e94-4f97-b610-67cde1cf001e" + ], + "confidence": "high" + }, + { + "claim": "The largest contributing segment was 'PR', responsible for 122.5% of the total change (evidence ID: e6ba043a-c288-49db-878a-a7f0720be4ce)", + "claim_type": "interpretation", + "evidence_ids": [ + "e6ba043a-c288-49db-878a-a7f0720be4ce" + ], + "confidence": "high" + } + ], + "limitations": [ + "The data does not specify the exact reasons for the change in average order value", + "The analysis does not account for external factors such as market trends or competitor activity" + ], + "recommended_next_checks": [ + "Inspect customer_state to understand if specific regions contributed to the average order value increase", + "Review product_category to identify if certain product types drove the revenue growth", + "Analyze seller performance to determine if specific sellers influenced the change in average order value" + ] + } + }, + { + "investigation_id": "69182461-d134-401a-a2d9-566f5125c352", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-01-01", + "end": "2018-01-31" + }, + "comparison_period": { + "start": "2017-12-01", + "end": "2017-12-31" + }, + "question": "Why did revenue decline?", + "status": "completed", + "step_count": 8, + "query_count": 6, + "cancel_requested": false, + "created_at": "2026-09-12T06:43:05.610388", + "updated_at": "2026-09-12T06:44:06.908283", + "report": { + "status": "partial", + "headline": "Product Revenue Growth Driven by SP Segment but Cancellation Rate Increase Raises Concerns", + "observed_change": { + "metric": "product_revenue", + "current_value": 945606.29, + "comparison_value": 742433.79, + "percent_change": 0.2736573991331941, + "evidence_ids": [ + "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9" + ] + }, + "findings": [ + { + "claim": "Product_revenue increased by 27.4% (evidence ID: 3be66378-ed39-4865-92f1-2e8a7c83bca7)", + "claim_type": "interpretation", + "evidence_ids": [ + "3be66378-ed39-4865-92f1-2e8a7c83bca7" + ], + "confidence": "high" + }, + { + "claim": "Order count increased by 27.8% (evidence ID: 6d536dc5-14a1-44e9-803a-3966949e6d3f)", + "claim_type": "interpretation", + "evidence_ids": [ + "6d536dc5-14a1-44e9-803a-3966949e6d3f" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate rose by 21.4% (evidence ID: 98c1f67a-6c49-4bd1-b295-65a46065d6b9)", + "claim_type": "interpretation", + "evidence_ids": [ + "98c1f67a-6c49-4bd1-b295-65a46065d6b9" + ], + "confidence": "high" + }, + { + "claim": "The SP segment contributed 46.4% of the total change (evidence ID: c52f6ec6-0cca-48bf-8f61-321e23c9c9d9)", + "claim_type": "interpretation", + "evidence_ids": [ + "c52f6ec6-0cca-48bf-8f61-321e23c9c9d9" + ], + "confidence": "high" + }, + { + "claim": "The cancellation rate increase suggests that revenue growth may be partially offset by cancellations (evidence ID: 98c1f67a-6c49-4bd1-b295-65a46065d6b9)", + "claim_type": "interpretation", + "evidence_ids": [ + "98c1f67a-6c49-4bd1-b295-65a46065d6b9" + ], + "confidence": "medium" + } + ], + "limitations": [ + "The hypothesis is inconclusive due to limited data on the relationship between cancellation rates and revenue growth.", + "The analysis does not account for external factors such as market trends or competitor activity.", + "The contribution of the SP segment to revenue growth is not fully validated without further analysis." + ], + "recommended_next_checks": [ + "Investigate the correlation between cancellation rates and revenue growth using detailed transaction data.", + "Analyze the SP segment's performance to validate its contribution to revenue growth.", + "Conduct a deeper analysis of the reasons behind the increased cancellation rate." + ] + } + }, + { + "investigation_id": "17f4ecd2-a471-4b18-84ab-911d90b7fe70", + "metric": "product_revenue", + "metric_definition_version": "product_revenue:v1", + "current_period": { + "start": "2018-02-01", + "end": "2018-02-28" + }, + "comparison_period": { + "start": "2018-01-01", + "end": "2018-01-31" + }, + "question": null, + "status": "completed", + "step_count": 5, + "query_count": 4, + "cancel_requested": false, + "created_at": "2026-09-12T06:31:56.516588", + "updated_at": "2026-09-12T06:32:36.325011", + "report": { + "status": "answered", + "headline": "Product Revenue Decline Driven by Cancellations in SP Segment", + "observed_change": { + "metric": "product_revenue", + "current_value": 837895.43, + "comparison_value": 945606.29, + "percent_change": -0.11390666616652897, + "evidence_ids": [ + "e2760221-e350-403f-a66b-a189bee1c1b2" + ] + }, + "findings": [ + { + "claim": "Product revenue decreased by 18.2% in the period, with the SP segment contributing 60.0% of the total decline.", + "claim_type": "interpretation", + "evidence_ids": [ + "00d6a3fb-019f-4952-bbfe-bc46bbdc7106" + ], + "confidence": "high" + }, + { + "claim": "Cancellation rate increased by 32.6% in the period, outpacing the 7.8% decrease in order count.", + "claim_type": "interpretation", + "evidence_ids": [], + "confidence": "high" + }, + { + "claim": "The SP segment accounted for 60.0% of the total change in product_revenue.", + "claim_type": "interpretation", + "evidence_ids": [ + "e2760221-e350-403f-a66b-a189bee1c1b2" + ], + "confidence": "high" + }, + { + "claim": "Order count decreased by 7.8%, while the cancellation rate increased by 32.6%.", + "claim_type": "interpretation", + "evidence_ids": [ + "ede03eec-8ac5-40b6-a940-3cd9c3b44779" + ], + "confidence": "high" + } + ], + "limitations": [ + "The analysis does not account for external factors such as market trends or competitor activity.", + "The data does not include information on customer behavior changes or product quality issues.", + "The relationship between cancellation rate and revenue decline is inferred but not directly measured." + ], + "recommended_next_checks": [ + "Review the SP segment's customer feedback and support tickets for patterns in cancellations.", + "Analyze the time series of cancellation rates and order counts to identify any seasonal trends or anomalies.", + "Conduct a cohort analysis of SP segment customers to assess retention and churn behavior over time." + ] + } + } +] \ No newline at end of file diff --git a/apps/web/public/demo/manifest.json b/apps/web/public/demo/manifest.json new file mode 100644 index 0000000..2bc003b --- /dev/null +++ b/apps/web/public/demo/manifest.json @@ -0,0 +1,90 @@ +[ + { + "key": "GET /api/metrics/summary?comparison_end=2017-12-31&comparison_start=2017-12-01¤t_end=2018-01-31¤t_start=2018-01-01", + "file": "metrics-summary-default.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352", + "file": "investigation-69182461-d134-401a-a2d9-566f5125c352.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352/events", + "file": "investigation-69182461-d134-401a-a2d9-566f5125c352-events.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352/evidence/c52f6ec6-0cca-48bf-8f61-321e23c9c9d9", + "file": "evidence-c52f6ec6-0cca-48bf-8f61-321e23c9c9d9.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352/evidence/3be66378-ed39-4865-92f1-2e8a7c83bca7", + "file": "evidence-3be66378-ed39-4865-92f1-2e8a7c83bca7.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352/evidence/6d536dc5-14a1-44e9-803a-3966949e6d3f", + "file": "evidence-6d536dc5-14a1-44e9-803a-3966949e6d3f.json" + }, + { + "key": "GET /api/investigations/69182461-d134-401a-a2d9-566f5125c352/evidence/98c1f67a-6c49-4bd1-b295-65a46065d6b9", + "file": "evidence-98c1f67a-6c49-4bd1-b295-65a46065d6b9.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70", + "file": "investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70/events", + "file": "investigation-17f4ecd2-a471-4b18-84ab-911d90b7fe70-events.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70/evidence/e2760221-e350-403f-a66b-a189bee1c1b2", + "file": "evidence-e2760221-e350-403f-a66b-a189bee1c1b2.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70/evidence/8b45bf68-ef1c-4ee8-a362-b5e0f0731221", + "file": "evidence-8b45bf68-ef1c-4ee8-a362-b5e0f0731221.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70/evidence/00d6a3fb-019f-4952-bbfe-bc46bbdc7106", + "file": "evidence-00d6a3fb-019f-4952-bbfe-bc46bbdc7106.json" + }, + { + "key": "GET /api/investigations/17f4ecd2-a471-4b18-84ab-911d90b7fe70/evidence/ede03eec-8ac5-40b6-a940-3cd9c3b44779", + "file": "evidence-ede03eec-8ac5-40b6-a940-3cd9c3b44779.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675", + "file": "investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675/events", + "file": "investigation-f0c3ba87-8f54-495a-b3e5-b8531e064675-events.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675/evidence/e6ba043a-c288-49db-878a-a7f0720be4ce", + "file": "evidence-e6ba043a-c288-49db-878a-a7f0720be4ce.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675/evidence/7764dd7b-f478-4bde-8610-64ca1e93dc35", + "file": "evidence-7764dd7b-f478-4bde-8610-64ca1e93dc35.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675/evidence/8595a0fb-6e94-4f97-b610-67cde1cf001e", + "file": "evidence-8595a0fb-6e94-4f97-b610-67cde1cf001e.json" + }, + { + "key": "GET /api/investigations/f0c3ba87-8f54-495a-b3e5-b8531e064675/evidence/0d5a34b0-4b86-48a9-aa49-887d234cda61", + "file": "evidence-0d5a34b0-4b86-48a9-aa49-887d234cda61.json" + }, + { + "key": "GET /api/investigations", + "file": "investigations-list.json" + }, + { + "key": "GET /api/evaluations", + "file": "evaluations-list.json" + }, + { + "key": "GET /api/evaluations/f51f692d-f106-430c-879f-dce4019e091a", + "file": "evaluation-f51f692d-f106-430c-879f-dce4019e091a.json" + } +] \ No newline at end of file diff --git a/apps/web/public/demo/metrics-summary-default.json b/apps/web/public/demo/metrics-summary-default.json new file mode 100644 index 0000000..880d340 --- /dev/null +++ b/apps/web/public/demo/metrics-summary-default.json @@ -0,0 +1,14 @@ +{ + "product_revenue": { + "current_value": 945606.29, + "comparison_value": 742433.79, + "absolute_change": 203172.5, + "percent_change": 0.2736573991331941 + }, + "orders": { + "current_value": 7189.0, + "comparison_value": 5623.0, + "absolute_change": 1566.0, + "percent_change": 0.27849902187444425 + } +} \ No newline at end of file diff --git a/docs/superpowers/plans/2026-09-11-hosted-demo.md b/docs/superpowers/plans/2026-09-11-hosted-demo.md new file mode 100644 index 0000000..4fcf710 --- /dev/null +++ b/docs/superpowers/plans/2026-09-11-hosted-demo.md @@ -0,0 +1,1036 @@ +# Hosted Read-Only Demo Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking. + +**Goal:** Deploy a $0, zero-maintenance, read-only demo of RootLens by capturing real +investigation runs to disk and serving them through a static export, so a recruiter +landing from the portfolio card sees a real, finished investigation with working +citations in under 5 seconds. + +**Architecture:** All frontend API calls already funnel through one function, +`fetchJson` in `apps/web/lib/api-client.ts`. In demo builds +(`NEXT_PUBLIC_DEMO_MODE=1`), that function delegates to a new fixture resolver +that fetches baked JSON from `public/demo/` instead of the network. A Python +capture script populates those fixtures from a real local run. The two dynamic +routes (`/investigations/[id]`, `/evaluations/[id]`) split into a thin server +`page.tsx` (for `generateStaticParams`) plus the existing client component, +because `output: "export"` requires static params and a `"use client"` page +cannot export them. + +**Tech Stack:** Next.js 15 (App Router, static export), TypeScript, Vitest, +Playwright, Python 3 stdlib (capture script), Vercel (hosting). + +**Spec:** `docs/superpowers/specs/2026-09-11-hosted-demo-design.md` + +## Global Constraints + +- **$0 hosting, zero ongoing maintenance.** Static export only — no server, no + database, no API keys in the deployed demo. +- **Read-only.** The investigate button and date pickers are disabled in demo + mode, not left live against a backend that doesn't exist. +- **Fixtures come from real local runs, never hand-authored JSON.** The SQL, + rows, and citations shown must be what RootLens actually produced. +- **A visible banner names the capture date and the reason** (the loop needs a + local LLM, so this replays real runs instead of computing new ones). +- **The capture script is not wired into CI** — it needs a local Ollama model, + and a CI job that cannot run is worse than no job. +- **Existing suites must stay green throughout:** 141 backend tests at the + `--cov-fail-under=91` gate, 22 frontend tests, `npm run typecheck`, `npm run + lint`. +- **A resolver miss must fail loudly**, never render an empty panel — an empty + panel reads as a bug in RootLens itself. + +--- + +### Task 1: Demo fixture resolver + +**Files:** +- Create: `apps/web/lib/demo-data.ts` +- Test: `apps/web/lib/__tests__/demo-data.test.ts` + +**Interfaces:** +- Produces: `buildDemoKey(method: string, path: string, params?: Record): string` + and `resolveDemoFixture(method: string, path: string, params?: Record): Promise` + — both exported from `apps/web/lib/demo-data.ts`. Task 2 imports + `resolveDemoFixture` from here. Task 3's capture script must build manifest + keys with the exact same rule as `buildDemoKey`: `"${method} ${path}"` with no + params, or `"${method} ${path}?${sortedParams.join("&")}"` where params are + sorted lexicographically by key and joined as `key=value`. + +- [ ] **Step 1: Write the failing tests** + +```typescript +// apps/web/lib/__tests__/demo-data.test.ts +import { afterEach, describe, expect, it, vi } from "vitest"; +import { buildDemoKey, resolveDemoFixture } from "@/lib/demo-data"; + +describe("buildDemoKey", () => { + it("returns a plain method-and-path key when there are no params", () => { + expect(buildDemoKey("GET", "/api/investigations")).toBe("GET /api/investigations"); + }); + + it("sorts params so callers passing them in a different order collide on the same key", () => { + const a = buildDemoKey("GET", "/api/metrics/summary", { + current_start: "2018-01-01", + comparison_start: "2017-12-01", + }); + const b = buildDemoKey("GET", "/api/metrics/summary", { + comparison_start: "2017-12-01", + current_start: "2018-01-01", + }); + expect(a).toBe(b); + expect(a).toBe( + "GET /api/metrics/summary?comparison_start=2017-12-01¤t_start=2018-01-01", + ); + }); +}); + +describe("resolveDemoFixture", () => { + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it("fetches the manifest, then the matching fixture file", async () => { + const manifest = [{ key: "GET /api/investigations", file: "investigations-list.json" }]; + const fixture = [{ investigation_id: "inv-1", status: "completed" }]; + + vi.stubGlobal( + "fetch", + vi.fn((url: string) => { + if (url === "/demo/manifest.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(manifest) }); + } + if (url === "/demo/investigations-list.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(fixture) }); + } + throw new Error(`unexpected fetch: ${url}`); + }), + ); + + const result = await resolveDemoFixture("GET", "/api/investigations"); + expect(result).toEqual(fixture); + }); + + it("throws when no manifest entry matches the request", async () => { + vi.stubGlobal( + "fetch", + vi.fn().mockResolvedValue({ ok: true, json: () => Promise.resolve([]) }), + ); + + await expect(resolveDemoFixture("GET", "/api/investigations")).rejects.toThrow( + "Demo fixture not found for GET /api/investigations", + ); + }); + + it("throws when the manifest itself fails to load", async () => { + vi.stubGlobal("fetch", vi.fn().mockResolvedValue({ ok: false, status: 500 })); + + await expect(resolveDemoFixture("GET", "/api/investigations")).rejects.toThrow( + "Failed to load /demo/manifest.json: 500", + ); + }); +}); +``` + +- [ ] **Step 2: Run the tests to verify they fail** + +Run: `cd apps/web && npx vitest run lib/__tests__/demo-data.test.ts` +Expected: FAIL — `Cannot find module '@/lib/demo-data'` (the file doesn't exist yet). + +- [ ] **Step 3: Write the resolver** + +```typescript +// apps/web/lib/demo-data.ts + +interface DemoManifestEntry { + key: string; + file: string; +} + +export function buildDemoKey( + method: string, + path: string, + params?: Record, +): string { + if (!params || Object.keys(params).length === 0) { + return `${method} ${path}`; + } + const query = Object.keys(params) + .sort() + .map((key) => `${key}=${params[key]}`) + .join("&"); + return `${method} ${path}?${query}`; +} + +async function fetchDemoJson(url: string): Promise { + const response = await fetch(url); + if (!response.ok) { + throw new Error(`Failed to load ${url}: ${response.status}`); + } + return response.json() as Promise; +} + +export async function resolveDemoFixture( + method: string, + path: string, + params?: Record, +): Promise { + const key = buildDemoKey(method, path, params); + const manifest = await fetchDemoJson("/demo/manifest.json"); + const entry = manifest.find((item) => item.key === key); + if (!entry) { + throw new Error(`Demo fixture not found for ${key}`); + } + return fetchDemoJson(`/demo/${entry.file}`); +} +``` + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: `cd apps/web && npx vitest run lib/__tests__/demo-data.test.ts` +Expected: PASS (4 tests) + +- [ ] **Step 5: Commit** + +```bash +git add apps/web/lib/demo-data.ts apps/web/lib/__tests__/demo-data.test.ts +git commit -m "feat(web): add the demo fixture resolver" +``` + +--- + +### Task 2: Wire the resolver into `fetchJson`, gate the build + +**Files:** +- Modify: `apps/web/lib/api-client.ts:139-156` (the `fetchJson` function) +- Modify: `apps/web/lib/__tests__/api-client.test.ts` (imports + one new `describe` block) +- Modify: `apps/web/next.config.ts` +- Modify: `apps/web/package.json` (`scripts`) + +**Interfaces:** +- Consumes: `resolveDemoFixture` from Task 1 (`apps/web/lib/demo-data.ts`). +- Produces: `NEXT_PUBLIC_DEMO_MODE` as the build-time flag every later task + reads (`process.env.NEXT_PUBLIC_DEMO_MODE`), and the `npm run build:demo` + script Task 4, 6, and 7 all invoke to produce the static export. + +- [ ] **Step 1: Write the failing test** + +Add `getInvestigations` to the existing import block at the top of +`apps/web/lib/__tests__/api-client.test.ts`: + +```typescript +import { + cancelInvestigation, + getEvaluation, + getEvaluations, + getEvidence, + getInvestigation, + getInvestigationEvents, + getInvestigations, + getMetricsSummary, +} from "@/lib/api-client"; +``` + +Append this block at the end of the file: + +```typescript +describe("fetchJson in demo mode", () => { + afterEach(() => { + vi.unstubAllGlobals(); + delete process.env.NEXT_PUBLIC_DEMO_MODE; + }); + + it("resolves from the demo fixture manifest instead of calling the real API", async () => { + process.env.NEXT_PUBLIC_DEMO_MODE = "1"; + const manifest = [{ key: "GET /api/investigations", file: "investigations-list.json" }]; + const fixture = [{ investigation_id: "inv-1", status: "completed" }]; + + vi.stubGlobal( + "fetch", + vi.fn((url: string) => { + if (url === "/demo/manifest.json") { + return Promise.resolve({ ok: true, json: () => Promise.resolve(manifest) }); + } + return Promise.resolve({ ok: true, json: () => Promise.resolve(fixture) }); + }), + ); + + const result = await getInvestigations(); + expect(result).toEqual(fixture); + }); +}); +``` + +- [ ] **Step 2: Run the test to verify it fails** + +Run: `cd apps/web && npx vitest run lib/__tests__/api-client.test.ts` +Expected: FAIL — the demo-mode test receives `undefined` (or throws on the +`API_BASE_URL`-constructed URL), because `fetchJson` doesn't check the flag yet. + +- [ ] **Step 3: Add the branch to `fetchJson`** + +In `apps/web/lib/api-client.ts`, add the import near the top: + +```typescript +import { resolveDemoFixture } from "@/lib/demo-data"; +``` + +Then at the start of the `fetchJson` function body (immediately after the +opening `{`, before `const url = new URL(...)`): + +```typescript + if (process.env.NEXT_PUBLIC_DEMO_MODE) { + return resolveDemoFixture(init?.method ?? "GET", path, params); + } +``` + +- [ ] **Step 4: Run the tests to verify they pass** + +Run: `cd apps/web && npx vitest run lib/__tests__/api-client.test.ts` +Expected: PASS (all existing tests plus the new one) + +- [ ] **Step 5: Gate the static export on the same flag** + +Replace the contents of `apps/web/next.config.ts`: + +```typescript +import type { NextConfig } from "next"; + +const nextConfig: NextConfig = { + ...(process.env.NEXT_PUBLIC_DEMO_MODE ? { output: "export" } : {}), +}; + +export default nextConfig; +``` + +- [ ] **Step 6: Add the demo build script** + +In `apps/web/package.json`, add to `"scripts"` (alongside the existing `"build"` line): + +```json + "build:demo": "NEXT_PUBLIC_DEMO_MODE=1 next build --turbopack", +``` + +- [ ] **Step 7: Run the full frontend suite and confirm the normal build is untouched** + +Run: `cd apps/web && npm run test && npm run typecheck` +Expected: PASS. Then confirm the flag is off by default: +Run: `cd apps/web && npm run build` +Expected: succeeds exactly as before (no `output: "export"`, since +`NEXT_PUBLIC_DEMO_MODE` is unset). + +- [ ] **Step 8: Commit** + +```bash +git add apps/web/lib/api-client.ts apps/web/lib/__tests__/api-client.test.ts \ + apps/web/next.config.ts apps/web/package.json +git commit -m "feat(web): route fetchJson through demo fixtures when NEXT_PUBLIC_DEMO_MODE is set" +``` + +--- + +### Task 3: Capture script and real fixtures + +**Files:** +- Create: `scripts/capture_demo.py` +- Create (generated by running the script, then committed): + `apps/web/public/demo/manifest.json`, + `apps/web/public/demo/investigation-ids.json`, + `apps/web/public/demo/evaluation-ids.json`, + `apps/web/public/demo/captured-at.json`, + and one JSON file per captured response. + +**Interfaces:** +- Produces: `public/demo/manifest.json` (array of `{key, file}`, using the + exact key format from Task 1's `buildDemoKey`); `public/demo/investigation-ids.json` + and `public/demo/evaluation-ids.json` (each a JSON array of id strings) — + Task 4's `generateStaticParams` reads these two directly. + `public/demo/captured-at.json` (`{"captured_at": "YYYY-MM-DD"}`) — Task 6's + banner reads this. + +- [ ] **Step 1: Write the capture script** + +```python +#!/usr/bin/env python3 +"""Captures real RootLens API responses into apps/web/public/demo/ so the +static demo build (`npm run build:demo`) can resolve requests from disk +instead of a live backend. + +Preconditions — a fully running local stack with a real local LLM: + make up && make migrate && make ingest-fixtures + ollama pull qwen3:8b # if not already pulled, then `ollama serve` + +Optionally, to also capture a benchmark run for the /evaluations page: + make eval + +Then, from the repo root: + python3 scripts/capture_demo.py + +Not wired into CI: it needs a local model, and a CI job that cannot run +is worse than no job. +""" + +import json +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from datetime import date +from pathlib import Path +from typing import Any + +API_BASE_URL = "http://localhost:8000" +OUT_DIR = Path(__file__).resolve().parent.parent / "apps/web/public/demo" + +INVESTIGATIONS: list[dict[str, Any]] = [ + { + "metric": "product_revenue", + "current_period": {"start": "2018-01-01", "end": "2018-01-31"}, + "comparison_period": {"start": "2017-12-01", "end": "2017-12-31"}, + "question": "Why did revenue decline?", + }, + { + "metric": "product_revenue", + "current_period": {"start": "2018-02-01", "end": "2018-02-28"}, + "comparison_period": {"start": "2018-01-01", "end": "2018-01-31"}, + "question": None, + }, + { + "metric": "orders", + "current_period": {"start": "2018-03-01", "end": "2018-03-31"}, + "comparison_period": {"start": "2018-02-01", "end": "2018-02-28"}, + "question": "What drove the change in order volume?", + }, +] + + +def build_demo_key(method: str, path: str, params: dict[str, str] | None = None) -> str: + """Must match apps/web/lib/demo-data.ts's buildDemoKey exactly.""" + if not params: + return f"{method} {path}" + query = "&".join(f"{k}={v}" for k, v in sorted(params.items())) + return f"{method} {path}?{query}" + + +def request( + method: str, + path: str, + body: dict[str, Any] | None = None, + query: dict[str, str] | None = None, +) -> Any: + url = f"{API_BASE_URL}{path}" + if query: + url = f"{url}?{urllib.parse.urlencode(sorted(query.items()))}" + data = json.dumps(body).encode() if body is not None else None + req = urllib.request.Request( + url, + data=data, + method=method, + headers={"Content-Type": "application/json"} if data else {}, + ) + with urllib.request.urlopen(req, timeout=30) as response: + return json.loads(response.read()) + + +def wait_for_api() -> None: + for _ in range(30): + try: + request("GET", "/api/health") + return + except (urllib.error.URLError, ConnectionError): + time.sleep(1) + print("API not reachable at http://localhost:8000 -- run `make up` first.", file=sys.stderr) + sys.exit(1) + + +def wait_for_completion(investigation_id: str) -> dict[str, Any]: + for _ in range(150): # up to 5 minutes + investigation = request("GET", f"/api/investigations/{investigation_id}") + if investigation["status"] != "running": + return investigation + time.sleep(2) + raise RuntimeError(f"investigation {investigation_id} never left 'running'") + + +def evidence_ids_from(events: list[dict[str, Any]]) -> list[str]: + ids = [] + for event in events: + if event["event_type"] == "tool_call": + evidence_id = event["payload"].get("evidence_id") + if evidence_id: + ids.append(evidence_id) + return ids + + +def write(relative_path: str, payload: Any) -> str: + file_path = OUT_DIR / relative_path + file_path.parent.mkdir(parents=True, exist_ok=True) + file_path.write_text(json.dumps(payload, indent=2)) + return relative_path + + +def main() -> None: + wait_for_api() + OUT_DIR.mkdir(parents=True, exist_ok=True) + + manifest: list[dict[str, str]] = [] + investigation_ids: list[str] = [] + + # The home page requests this on every load, with these exact + # hardcoded defaults (apps/web/app/page.tsx). + summary_params = { + "current_start": "2018-01-01", + "current_end": "2018-01-31", + "comparison_start": "2017-12-01", + "comparison_end": "2017-12-31", + } + summary = request("GET", "/api/metrics/summary", query=summary_params) + file_name = write("metrics-summary-default.json", summary) + manifest.append( + {"key": build_demo_key("GET", "/api/metrics/summary", summary_params), "file": file_name} + ) + + for spec in INVESTIGATIONS: + created = request("POST", "/api/investigations", spec) + investigation_id = created["investigation_id"] + print(f"created investigation {investigation_id}, waiting for completion...") + investigation = wait_for_completion(investigation_id) + investigation_ids.append(investigation_id) + + file_name = write(f"investigation-{investigation_id}.json", investigation) + manifest.append( + {"key": build_demo_key("GET", f"/api/investigations/{investigation_id}"), "file": file_name} + ) + + events = request("GET", f"/api/investigations/{investigation_id}/events") + file_name = write(f"investigation-{investigation_id}-events.json", events) + manifest.append( + { + "key": build_demo_key("GET", f"/api/investigations/{investigation_id}/events"), + "file": file_name, + } + ) + + for evidence_id in evidence_ids_from(events): + evidence = request( + "GET", f"/api/investigations/{investigation_id}/evidence/{evidence_id}" + ) + file_name = write(f"evidence-{evidence_id}.json", evidence) + manifest.append( + { + "key": build_demo_key( + "GET", f"/api/investigations/{investigation_id}/evidence/{evidence_id}" + ), + "file": file_name, + } + ) + + investigations_list = request("GET", "/api/investigations") + file_name = write("investigations-list.json", investigations_list) + manifest.append({"key": build_demo_key("GET", "/api/investigations"), "file": file_name}) + + evaluation_ids: list[str] = [] + evaluations_list = request("GET", "/api/evaluations") + file_name = write("evaluations-list.json", evaluations_list) + manifest.append({"key": build_demo_key("GET", "/api/evaluations"), "file": file_name}) + + if evaluations_list: + run_id = evaluations_list[0]["run_id"] + evaluation_ids.append(run_id) + evaluation_detail = request("GET", f"/api/evaluations/{run_id}") + file_name = write(f"evaluation-{run_id}.json", evaluation_detail) + manifest.append( + {"key": build_demo_key("GET", f"/api/evaluations/{run_id}"), "file": file_name} + ) + else: + print( + "no evaluation runs found -- run `make eval` first if you want the " + "benchmark page in the demo", + file=sys.stderr, + ) + + write("manifest.json", manifest) + write("investigation-ids.json", investigation_ids) + write("evaluation-ids.json", evaluation_ids) + write("captured-at.json", {"captured_at": date.today().isoformat()}) + + print( + f"captured {len(investigation_ids)} investigations, " + f"{len(evaluation_ids)} evaluation run(s) -> {OUT_DIR}" + ) + + +if __name__ == "__main__": + main() +``` + +- [ ] **Step 2: Bring up the local stack** + +Run: `make up && make migrate && make ingest-fixtures` +Expected: all three services healthy; `curl -s localhost:8000/api/health` returns +`{"status":"ok"}` (or equivalent 200). + +Confirm Ollama is reachable and has the model: `ollama list` should show +`qwen3:8b`. If not: `ollama pull qwen3:8b`. + +- [ ] **Step 3: Run the capture script** + +Run: `python3 scripts/capture_demo.py` +Expected: prints "created investigation ..." three times, then "captured 3 +investigations, N evaluation run(s) -> .../apps/web/public/demo". This can +take several minutes — each investigation runs the real bounded loop against +the local model. + +If you want the `/evaluations` page in the demo too, run `make eval` before +this step so `evaluations_list` is non-empty. + +- [ ] **Step 4: Sanity-check the output** + +Run: `python3 -m json.tool apps/web/public/demo/manifest.json | head -20` +Expected: valid JSON, at least 8 entries (1 summary + 3 investigations + 3 +event lists + 1 investigations list, plus one evidence entry per exhibit). + +Run: `cat apps/web/public/demo/investigation-ids.json` +Expected: a JSON array of exactly 3 UUID strings. + +- [ ] **Step 5: Commit** + +```bash +git add scripts/capture_demo.py apps/web/public/demo +git commit -m "feat: capture real investigation runs for the static demo" +``` + +--- + +### Task 4: Split the two dynamic routes for static export + +**Files:** +- Create: `apps/web/app/investigations/[id]/investigation-view.tsx` + (the current body of `page.tsx`, as a named-export client component) +- Modify: `apps/web/app/investigations/[id]/page.tsx` (replaced with a thin + server component) +- Create: `apps/web/app/evaluations/[id]/evaluation-view.tsx` +- Modify: `apps/web/app/evaluations/[id]/page.tsx` + +**Interfaces:** +- Consumes: `apps/web/public/demo/investigation-ids.json` and + `apps/web/public/demo/evaluation-ids.json` from Task 3. +- Produces: `InvestigationView({ id }: { id: string })` and + `EvaluationView({ id }: { id: string })`, both **named** exports, imported + by their respective server `page.tsx`. + +- [ ] **Step 1: Create `investigation-view.tsx` from the current page** + +Copy the entire current contents of +`apps/web/app/investigations/[id]/page.tsx` into a new file +`apps/web/app/investigations/[id]/investigation-view.tsx`, then make two +changes to the copy: + +1. Remove `use` from the React import (no longer needed): + ```typescript + import { useCallback, useEffect, useMemo, useRef, useState } from "react"; + ``` +2. Replace the component signature and drop the `use(params)` line: + ```typescript + export function InvestigationView({ id }: { id: string }) { + ``` + (delete the old `const { id } = use(params);` line — `id` now arrives as a prop) + +Everything else in the file — all state, `load`, polling, `openEvidence`, +`handleCancel`, the JSX — is unchanged. + +- [ ] **Step 2: Replace `page.tsx` with a thin server component** + +```typescript +// apps/web/app/investigations/[id]/page.tsx +import fs from "node:fs"; +import path from "node:path"; +import { InvestigationView } from "./investigation-view"; + +export function generateStaticParams() { + if (!process.env.NEXT_PUBLIC_DEMO_MODE) return []; + const idsPath = path.join(process.cwd(), "public/demo/investigation-ids.json"); + const ids = JSON.parse(fs.readFileSync(idsPath, "utf-8")) as string[]; + return ids.map((id) => ({ id })); +} + +export default async function InvestigationPage({ + params, +}: { + params: Promise<{ id: string }>; +}) { + const { id } = await params; + return ; +} +``` + +- [ ] **Step 3: Repeat the same split for evaluations** + +Copy `apps/web/app/evaluations/[id]/page.tsx` into +`apps/web/app/evaluations/[id]/evaluation-view.tsx`, then: + +1. Remove `use` from the React import: + ```typescript + import { useEffect, useMemo, useState } from "react"; + ``` +2. Change the signature and drop the `use(params)` line: + ```typescript + export function EvaluationView({ id }: { id: string }) { + ``` + +Then replace `apps/web/app/evaluations/[id]/page.tsx`: + +```typescript +// apps/web/app/evaluations/[id]/page.tsx +import fs from "node:fs"; +import path from "node:path"; +import { EvaluationView } from "./evaluation-view"; + +export function generateStaticParams() { + if (!process.env.NEXT_PUBLIC_DEMO_MODE) return []; + const idsPath = path.join(process.cwd(), "public/demo/evaluation-ids.json"); + const ids = JSON.parse(fs.readFileSync(idsPath, "utf-8")) as string[]; + return ids.map((id) => ({ id })); +} + +export default async function EvaluationDetailPage({ + params, +}: { + params: Promise<{ id: string }>; +}) { + const { id } = await params; + return ; +} +``` + +- [ ] **Step 4: Typecheck and confirm no behavior regression** + +Run: `cd apps/web && npm run typecheck` +Expected: PASS (no unused `use` import, no signature mismatches). + +Run: `cd apps/web && npm run test:e2e` +Expected: PASS — `investigation.spec.ts` and `evaluations.spec.ts` exercise +these exact pages via `npm run dev` (normal, non-demo mode) and must still pass +unchanged, since the rendered behavior is identical. + +- [ ] **Step 5: Confirm the demo build actually produces static pages for the captured ids** + +Run: `cd apps/web && npm run build:demo` +Expected: succeeds, and the build log lists one static route per id from +`investigation-ids.json` and `evaluation-ids.json` (e.g. +`○ /investigations/`), not a single dynamic `[id]` entry. + +- [ ] **Step 6: Commit** + +```bash +git add apps/web/app/investigations/\[id\] apps/web/app/evaluations/\[id\] +git commit -m "refactor(web): split dynamic routes so they can be statically exported" +``` + +--- + +### Task 5: Demo banner and disabled controls + +**Files:** +- Create: `apps/web/components/demo-banner.tsx` +- Modify: `apps/web/app/layout.tsx` +- Modify: `apps/web/app/page.tsx` + +**Interfaces:** +- Consumes: `apps/web/public/demo/captured-at.json` from Task 3. + +- [ ] **Step 1: Write the banner component** + +```typescript +// apps/web/components/demo-banner.tsx +import fs from "node:fs"; +import path from "node:path"; + +export function DemoBanner() { + const metaPath = path.join(process.cwd(), "public/demo/captured-at.json"); + const { captured_at: capturedAt } = JSON.parse(fs.readFileSync(metaPath, "utf-8")) as { + captured_at: string; + }; + + return ( +
+ Read-only demo — investigations captured {capturedAt}. The investigation + loop needs a local LLM, so this replays real runs instead of computing + new ones. +
+ ); +} +``` + +- [ ] **Step 2: Render it conditionally from the layout** + +In `apps/web/app/layout.tsx`, add the import: + +```typescript +import { DemoBanner } from "@/components/demo-banner"; +``` + +Then, as the first child inside ``, before the +`
`: + +```tsx +{process.env.NEXT_PUBLIC_DEMO_MODE && } +``` + +- [ ] **Step 3: Disable the date pickers and investigate button in demo mode** + +In `apps/web/app/page.tsx`: + +Give `DateField` a `disabled` prop: + +```typescript +function DateField({ + label, + value, + onChange, + disabled, +}: { + label: string; + value: string; + onChange: (value: string) => void; + disabled?: boolean; +}) { + return ( + + ); +} +``` + +Inside `Home()`, add near the top of the function body: + +```typescript + const isDemoMode = process.env.NEXT_PUBLIC_DEMO_MODE === "1"; +``` + +Pass `disabled={isDemoMode}` to all four `` call sites (current +start/end, comparison start/end). + +Change the investigate button's `disabled` prop: + +```typescript + disabled={isInvestigating || isDemoMode} +``` + +And add, immediately after the existing `{investigateError && (...)}` block: + +```tsx + {isDemoMode && ( +

+ Disabled in this read-only demo. +

+ )} +``` + +- [ ] **Step 4: Confirm normal mode is unaffected** + +Run: `cd apps/web && npm run test && npm run typecheck` +Expected: PASS. `isDemoMode` is `false` when the env var is unset, so every +existing behavior and test is unchanged. + +- [ ] **Step 5: Commit** + +```bash +git add apps/web/components/demo-banner.tsx apps/web/app/layout.tsx apps/web/app/page.tsx +git commit -m "feat(web): add the read-only demo banner and disable controls in demo mode" +``` + +--- + +### Task 6: End-to-end test against the built static export + +**Files:** +- Create: `apps/web/playwright.demo.config.ts` +- Create: `apps/web/e2e/demo.spec.ts` +- Modify: `apps/web/package.json` (`scripts`) + +**Interfaces:** +- Consumes: the `out/` directory produced by `npm run build:demo` (Task 2, 4). + +- [ ] **Step 1: Write the demo Playwright config** + +```typescript +// apps/web/playwright.demo.config.ts +import { defineConfig, devices } from "@playwright/test"; + +export default defineConfig({ + testDir: "./e2e", + testMatch: "demo.spec.ts", + fullyParallel: true, + forbidOnly: !!process.env.CI, + retries: process.env.CI ? 2 : 0, + reporter: "list", + use: { + baseURL: "http://localhost:4173", + trace: "retain-on-failure", + }, + webServer: { + command: "npx --yes serve@latest out -l 4173", + url: "http://localhost:4173", + reuseExistingServer: !process.env.CI, + timeout: 60_000, + }, + projects: [{ name: "chromium", use: { ...devices["Desktop Chrome"] } }], +}); +``` + +- [ ] **Step 2: Write the demo spec** + +```typescript +// apps/web/e2e/demo.spec.ts +import { expect, test } from "@playwright/test"; + +test("landing investigation renders with a working citation, and controls are disabled", async ({ + page, +}) => { + await page.goto("/"); + + await expect(page.getByText(/Read-only demo/)).toBeVisible(); + await expect( + page.getByRole("button", { name: "Investigate revenue change" }), + ).toBeDisabled(); + + await page.locator('a[href^="/investigations/"]').first().click(); + + await expect(page.getByTestId("investigation-status")).toHaveText("completed"); + + const firstCitation = page.getByRole("button", { name: /^E1/ }); + await firstCitation.click(); + + const dialog = page.getByRole("dialog", { name: "Evidence detail" }); + await expect(dialog).toBeVisible(); + await expect(dialog.getByText("rows")).toBeVisible(); +}); +``` + +- [ ] **Step 3: Add the npm script** + +In `apps/web/package.json`, add to `"scripts"`: + +```json + "test:e2e:demo": "playwright test --config=playwright.demo.config.ts", +``` + +- [ ] **Step 4: Build the demo export and run the spec** + +Run: `cd apps/web && npm run build:demo` +Expected: succeeds, produces `apps/web/out/`. + +Run: `cd apps/web && npm run test:e2e:demo` +Expected: PASS. This is the test that catches drift between the demo and the +real app — it exercises the same `InvestigationView` and `EvidenceDrawer` +components the live app uses. + +If it fails on the citation step, check that at least one captured +investigation's report actually cites evidence (`findings[].evidence_ids` +non-empty) — rerun `scripts/capture_demo.py` against a scenario that produces +findings if not. + +- [ ] **Step 5: Commit** + +```bash +git add apps/web/playwright.demo.config.ts apps/web/e2e/demo.spec.ts apps/web/package.json +git commit -m "test(web): add an e2e spec against the built demo export" +``` + +--- + +### Task 7: Deploy and link from the portfolio (manual — external account) + +This task creates a new project on an external service tied to your account. +It is not something to run unsupervised — walk through it yourself, or have +it run with you watching. + +**Files:** +- Modify (in the Portfolio repo, `~/Desktop/website/Portfolio`): + `src/data/projects.tsx` (RootLens entry, add a demo URL) + +- [ ] **Step 1: Push the `hosted-demo` branch and open a PR** + +```bash +git push -u origin hosted-demo +gh pr create --title "Add a hosted read-only demo" --body "$(cat <<'EOF' +## Summary +- Static, read-only demo of RootLens, built from real captured investigation runs +- No server, no database, no API keys in the deployed site — $0, zero maintenance +- See docs/superpowers/specs/2026-09-11-hosted-demo-design.md for the design + +## Test plan +- [x] apps/web unit tests pass (`npm run test`) +- [x] apps/web e2e against the built static export (`npm run test:e2e:demo`) +- [x] Normal (non-demo) build unaffected (`npm run build`) +EOF +)" +``` + +Wait for CI to go green, then merge (same flow as the `coverage-reporting` +PR: `gh pr merge --squash` or via the GitHub UI, whichever you prefer). + +- [ ] **Step 2: Create the Vercel project** + +From `apps/web/`, with the Vercel CLI (`npm i -g vercel` if you don't have +it): + +```bash +cd apps/web +vercel login # interactive — opens a browser +vercel link # create a NEW project, separate from the portfolio's Vercel project +``` + +When prompted for build settings: +- Build command: `npm run build:demo` +- Output directory: `out` + +- [ ] **Step 3: Deploy to production** + +```bash +vercel --prod +``` + +Expected: prints a `https://.vercel.app` URL. Open it and confirm the +banner, the landing investigation, and a citation click all work exactly as +in the local `test:e2e:demo` run. + +- [ ] **Step 4: Link it from the portfolio** + +In `~/Desktop/website/Portfolio/src/data/projects.tsx`, find the RootLens +entry and add the deployed URL as its demo link (following whatever field the +`Project` type already uses for a live-demo link — check the type definition +in that file before adding a new one). + +- [ ] **Step 5: Commit and open a PR in the Portfolio repo** + +```bash +cd ~/Desktop/website/Portfolio +git checkout -b link-rootlens-demo +git add src/data/projects.tsx +git commit -m "Link the RootLens hosted demo from its project card" +git push -u origin link-rootlens-demo +gh pr create --title "Link the RootLens hosted demo" --body "Adds the deployed static demo URL to the RootLens project card." +``` + +--- + +## Task Order and Dependencies + +Tasks 1 → 2 → 3 are strictly sequential (each produces an interface the next +consumes), and Task 4 must follow Task 3: its `generateStaticParams` reads +`investigation-ids.json` / `evaluation-ids.json`, and its Step 5 check ("one +static route per id") is only meaningful once those ids come from a real +capture. Task 5 depends on Task 3's `captured-at.json`. Task 6 depends on +Tasks 2, 4, and 5 all being in place (it builds and tests the +whole demo). Task 7 depends on everything before it being merged. diff --git a/docs/superpowers/specs/2026-09-11-hosted-demo-design.md b/docs/superpowers/specs/2026-09-11-hosted-demo-design.md new file mode 100644 index 0000000..0201dd6 --- /dev/null +++ b/docs/superpowers/specs/2026-09-11-hosted-demo-design.md @@ -0,0 +1,183 @@ +# Hosted read-only demo — design + +**Date:** 2026-09-11 +**Status:** Approved, not yet implemented +**Scope:** `apps/web`, plus one new capture script under `scripts/` + +## Problem + +RootLens is complete and CI-green, but there is nowhere to send someone who +wants to see it. The investigation loop calls a local Ollama model, and no free +hosting tier runs a local LLM, so the running app cannot simply be deployed. + +## Audience and constraints + +The demo targets recruiters and hiring managers arriving from the portfolio +card. That fixes three constraints: + +- **The visit is 60–90 seconds.** Proof must be on screen before any scrolling. + A free-tier backend that cold-starts for 30–60s is worse than no demo, because + the visit ends before the page wakes. +- **$0 and zero maintenance.** No server, no database, no API keys, no + rate-limiting, nothing that rots or can be abused. +- **Read-only.** The visitor lands on a finished investigation and can browse + the other two. They do not run their own. + +Non-goals: live investigations, a question box, a hosted LLM, any always-on +infrastructure. These are not deferred features to design around — they are +excluded. + +## Approach + +Snapshot-and-serve. Real investigations are run locally, every API response the +UI touches is captured to disk, and the built site resolves requests from those +files instead of the network. + +The frontend already funnels all nine API functions through one 20-line helper, +`fetchJson` in `apps/web/lib/api-client.ts:139`. That is the only seam the demo +needs. + +### Approaches rejected + +- **Point `NEXT_PUBLIC_API_BASE_URL` at static JSON files.** Does not work. Every + GET carries query parameters (`/api/metrics/summary?current_start=…`) and two + endpoints are POSTs, so request shapes do not map onto file paths. +- **A separate `/demo` route group** that imports fixtures and passes props to the + existing components. Leaves `api-client.ts` untouched, but duplicates the + page-level composition of three routes. The demo and the real pages then drift + apart silently, which is the failure mode that makes a demo misrepresent the + product without anyone noticing. +- **Monkeypatching `globalThis.fetch` in demo builds.** Nothing typed changes, + but failures are invisible and "did the patch apply?" becomes a permanent + debugging step. No payoff over the chosen approach. + +## Components + +### 1. Fixture resolver — `apps/web/lib/demo-data.ts` (new) + +Maps a request signature to a baked JSON file listed in +`public/demo/manifest.json`. No knowledge of React or of any endpoint's meaning. + +**Key format.** `METHOD /path?k=v&k2=v2`, with parameter keys sorted +lexicographically, so callers passing the same parameters in a different order +hit the same fixture. Example: +`GET /api/metrics/summary?comparison_end=…&comparison_start=…¤t_end=…¤t_start=…`. + +**Fixtures are fetched, not bundled.** They live under `public/demo/` and the +resolver requests them by same-origin relative URL (`/demo/.json`). The +alternative — importing them as modules — would inline every evidence row into +the JS bundle and slow the first paint, which is the one thing this audience +cannot afford. A same-origin static JSON request costs single-digit +milliseconds and is cacheable. + +The manifest itself is the only fixture bundled with the app, so the resolver +can answer "is this request part of the demo?" without a round trip. + +### 2. One branch in `fetchJson` — `apps/web/lib/api-client.ts` (modified) + +When `process.env.NEXT_PUBLIC_DEMO_MODE` is set, `fetchJson` delegates to the +resolver instead of building a URL against `API_BASE_URL`. Everything else in +the file — all nine endpoint functions, every exported type — is unchanged. + +The resolver still performs a `fetch`, but against a same-origin static file +rather than the API, so the built site has no backend dependency. + +The flag is read at build time, so a normal build produces the same bundle it +does today and the demo path is not reachable in it. + +### 3. Capture script — `scripts/capture_demo.py` (new) + +Runs against a live local stack (Postgres on 5433, Ollama `qwen3:8b`): + +1. Creates three investigations over the Olist dataset. +2. Polls each to a terminal status. +3. GETs every endpoint the UI touches for those investigations, plus the + evaluations list and one evaluation detail. +4. Writes each response to `apps/web/public/demo/` and emits `manifest.json`. + +Output is committed. The script is **not** wired into CI: it needs a local model, +and a CI job that cannot run is worse than no job. + +Fixtures come from real runs. Hand-authoring the JSON would be faster and would +quietly turn the demo into a mockup — the SQL, rows, and citations on screen must +be what RootLens actually produced. + +### 4. Static export route split — `apps/web/app/**` (modified) + +All four pages are `"use client"`. Under `output: "export"`, the two dynamic +routes must export `generateStaticParams`, and a client component cannot export +it. Each therefore splits in two: + +| Route | Today | After | +| --- | --- | --- | +| `/investigations/[id]` | `page.tsx`, 348 lines, client | server `page.tsx` (params + `generateStaticParams`) + `investigation-view.tsx` (client, the current body) | +| `/evaluations/[id]` | `page.tsx`, 171 lines, client | server `page.tsx` + `evaluation-view.tsx` (client) | + +`generateStaticParams` reads the ids from the demo manifest. + +This is the largest piece of the work. It is also a structure the app wants +independently of the demo, so it is not throwaway. + +### 5. Read-only affordances — `apps/web/app/page.tsx` (modified) + +In demo mode the investigate button and date pickers are disabled with a short +inline reason. They are not left live to fail against a backend that is not +there. + +A banner names the capture date and states that investigations are pre-run +because the loop needs a local LLM. This follows the project's existing +`publish-honest-metrics` principle: say what the number or the artifact actually +is, rather than letting a visitor infer something more flattering. + +### 6. `next.config.ts` (modified) + +Sets `output: "export"` only when `NEXT_PUBLIC_DEMO_MODE` is set. + +## Data flow + +``` +capture_demo.py ──> public/demo/*.json + manifest.json (build-time, manual) + │ +browser ──> page ──> api-client fn ──> fetchJson ──> demo-data resolver ──> JSON + │ + └─ (normal build) ──> fetch ──> FastAPI +``` + +## Error handling + +A resolver miss — a request the capture script never recorded — throws during +build and in dev, and renders a plain "not part of this demo" message in the +built site. + +It must never render an empty panel. An empty panel reads as a product bug in +RootLens itself, which is the opposite of what the demo exists to show. + +## Testing + +- **Resolver unit tests** in the existing `apps/web/lib/__tests__/api-client.test.ts`: + hit, miss, and parameter-order independence. +- **One Playwright spec** (`apps/web/e2e/` is already configured) run against the + *built static export*, asserting that the landing investigation renders and that + clicking a citation opens the evidence drawer. This is the test that catches + demo-versus-real drift, because it exercises the same components the live app + uses. +- Existing suites must stay green: 141 backend tests at ≥91% (the CI gate), 22 + frontend tests. + +## Deployment + +Vercel, as its own project separate from the portfolio. Static output, free tier, +no cold start. The portfolio's RootLens card links to it. + +## Build order + +1. Resolver + unit tests (no UI involved, testable in isolation). +2. `fetchJson` branch. +3. Capture script; produce real fixtures. +4. Route split for the two dynamic routes. +5. Demo affordances and banner. +6. Playwright spec against the built export. +7. Deploy; link from the portfolio card. + +Step 4 is the long pole. Steps 1–3 are independently verifiable before any page +is touched. diff --git a/scripts/capture_demo.py b/scripts/capture_demo.py new file mode 100644 index 0000000..5174a2d --- /dev/null +++ b/scripts/capture_demo.py @@ -0,0 +1,221 @@ +#!/usr/bin/env python3 +"""Captures real RootLens API responses into apps/web/public/demo/ so the +static demo build (`npm run build:demo`) can resolve requests from disk +instead of a live backend. + +Preconditions — a fully running local stack with a real local LLM: + make up && make migrate && make ingest-fixtures + ollama pull qwen3:8b # if not already pulled, then `ollama serve` + +Optionally, to also capture a benchmark run for the /evaluations page: + make eval + +Then, from the repo root: + python3 scripts/capture_demo.py + +Not wired into CI: it needs a local model, and a CI job that cannot run +is worse than no job. +""" + +import json +import os +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +from datetime import date +from pathlib import Path +from typing import Any + +API_BASE_URL = os.environ.get("ROOTLENS_API_URL", "http://localhost:8000") +OUT_DIR = Path(__file__).resolve().parent.parent / "apps/web/public/demo" + +INVESTIGATIONS: list[dict[str, Any]] = [ + { + "metric": "product_revenue", + "current_period": {"start": "2018-01-01", "end": "2018-01-31"}, + "comparison_period": {"start": "2017-12-01", "end": "2017-12-31"}, + "question": "Why did revenue decline?", + }, + { + "metric": "product_revenue", + "current_period": {"start": "2018-02-01", "end": "2018-02-28"}, + "comparison_period": {"start": "2018-01-01", "end": "2018-01-31"}, + "question": None, + }, + { + "metric": "product_revenue", + "current_period": {"start": "2018-04-01", "end": "2018-04-30"}, + "comparison_period": {"start": "2018-03-01", "end": "2018-03-31"}, + "question": "What drove the change in revenue this period?", + }, +] + + +def build_demo_key(method: str, path: str, params: dict[str, str] | None = None) -> str: + """Must match apps/web/lib/demo-data.ts's buildDemoKey exactly.""" + if not params: + return f"{method} {path}" + query = "&".join(f"{k}={v}" for k, v in sorted(params.items())) + return f"{method} {path}?{query}" + + +def request( + method: str, + path: str, + body: dict[str, Any] | None = None, + query: dict[str, str] | None = None, +) -> Any: + url = f"{API_BASE_URL}{path}" + if query: + url = f"{url}?{urllib.parse.urlencode(sorted(query.items()))}" + data = json.dumps(body).encode() if body is not None else None + req = urllib.request.Request( + url, + data=data, + method=method, + headers={"Content-Type": "application/json"} if data else {}, + ) + with urllib.request.urlopen(req, timeout=30) as response: + return json.loads(response.read()) + + +def wait_for_api() -> None: + for _ in range(30): + try: + request("GET", "/api/health") + return + except (urllib.error.URLError, ConnectionError): + time.sleep(1) + print(f"API not reachable at {API_BASE_URL} -- run `make up` first.", file=sys.stderr) + sys.exit(1) + + +def wait_for_completion(investigation_id: str) -> dict[str, Any]: + for _ in range(150): # up to 5 minutes + investigation = request("GET", f"/api/investigations/{investigation_id}") + if investigation["status"] != "running": + return investigation + time.sleep(2) + raise RuntimeError(f"investigation {investigation_id} never left 'running'") + + +def evidence_ids_from(events: list[dict[str, Any]]) -> list[str]: + ids = [] + for event in events: + if event["event_type"] == "tool_call": + evidence_id = event["payload"].get("evidence_id") + if evidence_id: + ids.append(evidence_id) + return ids + + +def write(relative_path: str, payload: Any) -> str: + file_path = OUT_DIR / relative_path + file_path.parent.mkdir(parents=True, exist_ok=True) + file_path.write_text(json.dumps(payload, indent=2)) + return relative_path + + +def main() -> None: + wait_for_api() + OUT_DIR.mkdir(parents=True, exist_ok=True) + + manifest: list[dict[str, str]] = [] + investigation_ids: list[str] = [] + + # The home page requests this on every load, with these exact + # hardcoded defaults (apps/web/app/page.tsx). + summary_params = { + "current_start": "2018-01-01", + "current_end": "2018-01-31", + "comparison_start": "2017-12-01", + "comparison_end": "2017-12-31", + } + summary = request("GET", "/api/metrics/summary", query=summary_params) + file_name = write("metrics-summary-default.json", summary) + manifest.append( + {"key": build_demo_key("GET", "/api/metrics/summary", summary_params), "file": file_name} + ) + + for spec in INVESTIGATIONS: + created = request("POST", "/api/investigations", spec) + investigation_id = created["investigation_id"] + print(f"created investigation {investigation_id}, waiting for completion...") + investigation = wait_for_completion(investigation_id) + investigation_ids.append(investigation_id) + + file_name = write(f"investigation-{investigation_id}.json", investigation) + manifest.append( + {"key": build_demo_key("GET", f"/api/investigations/{investigation_id}"), "file": file_name} + ) + + events = request("GET", f"/api/investigations/{investigation_id}/events") + file_name = write(f"investigation-{investigation_id}-events.json", events) + manifest.append( + { + "key": build_demo_key( + "GET", f"/api/investigations/{investigation_id}/events" + ), + "file": file_name, + } + ) + + for evidence_id in evidence_ids_from(events): + evidence = request( + "GET", f"/api/investigations/{investigation_id}/evidence/{evidence_id}" + ) + file_name = write(f"evidence-{evidence_id}.json", evidence) + manifest.append( + { + "key": build_demo_key( + "GET", f"/api/investigations/{investigation_id}/evidence/{evidence_id}" + ), + "file": file_name, + } + ) + + investigations_list = request("GET", "/api/investigations") + investigations_list = [ + inv for inv in investigations_list if inv["investigation_id"] in investigation_ids + ] + file_name = write("investigations-list.json", investigations_list) + manifest.append({"key": build_demo_key("GET", "/api/investigations"), "file": file_name}) + + evaluation_ids: list[str] = [] + evaluations_list = request("GET", "/api/evaluations") + file_name = write("evaluations-list.json", evaluations_list) + manifest.append({"key": build_demo_key("GET", "/api/evaluations"), "file": file_name}) + + if evaluations_list: + held_out_run = next( + (run for run in evaluations_list if run.get("split") == "held_out"), None + ) + run_id = (held_out_run or evaluations_list[0])["run_id"] + evaluation_ids.append(run_id) + evaluation_detail = request("GET", f"/api/evaluations/{run_id}") + file_name = write(f"evaluation-{run_id}.json", evaluation_detail) + manifest.append( + {"key": build_demo_key("GET", f"/api/evaluations/{run_id}"), "file": file_name} + ) + else: + print( + "no evaluation runs found -- run `make eval` first if you want the " + "benchmark page in the demo", + file=sys.stderr, + ) + + write("manifest.json", manifest) + write("investigation-ids.json", investigation_ids) + write("evaluation-ids.json", evaluation_ids) + write("captured-at.json", {"captured_at": date.today().isoformat()}) + + print( + f"captured {len(investigation_ids)} investigations, " + f"{len(evaluation_ids)} evaluation run(s) -> {OUT_DIR}" + ) + + +if __name__ == "__main__": + main()